From 827499a30962e19d7245aeb55667501030cecd95 Mon Sep 17 00:00:00 2001 From: Yam ZhengLim Date: Mon, 26 Dec 2022 16:50:22 +0800 Subject: [PATCH] Debug CRISP naming --- __pycache__/general_function.cpython-39.pyc | Bin 1405 -> 1449 bytes .../crisp_extract_function.cpython-39.pyc | Bin 3655 -> 3738 bytes crisp/extract/crisp_extract_function.py | 22 ++++++++++-------- general_function.py | 5 ++-- 4 files changed, 15 insertions(+), 12 deletions(-) diff --git a/__pycache__/general_function.cpython-39.pyc b/__pycache__/general_function.cpython-39.pyc index 8e8548e42e5d03b2d6b9ecc0e39db594a1a98468..a0d8b9156e6bbd3034dd71de92512128b1cd2c6a 100644 GIT binary patch delta 282 zcmey%wUS#qk(ZZ?0SK1HuS`~AVPJR+;vfTNAjbiSi@hdlJMiAVXS4U5-DNGVyt1PVM<}_WvXSa zVGd^qW?*3`VXE|FWT<6fWT=uWVXk3mW@Kb2VNPKJ(cuh{3@JdlI+hgXU?|Ig!2rkt zI)TMclkpaFacaTjVir$P)*_%67>Z5Fg`gWv1sxk7Gqffpb!Hi3nLRFA0rzO fa{ytHJWx!Nu}EezC#yQ6++;&mVMh7MuB;XSr8F>Y delta 270 zcmZ3<{g+ESk(ZZ?0SJ6I6eoXYW?*;>;vfSiAjbiSi#;Z4JMeO*aJ4W*ai%Z^GiY*e ztkPg)bl$v}F_o!4nGs17&>RMksm>sMK0u;MpoC!oV-3SX##+W2#&CvU1{Q`Arb;(P zhFT^@hAO!d<{GAEMn;Ab<`hN{9nKKRkOGveV@hEPhO!J841g@4z07`9s?lzte$FAD z{(kYUK|%gOv6E-An6oh#u>kFw{FKF5gb&DIU}R$CVq|0FW8?tBBH78Ntm>jNK=C4w QY>^mumAu6 diff --git a/crisp/extract/__pycache__/crisp_extract_function.cpython-39.pyc b/crisp/extract/__pycache__/crisp_extract_function.cpython-39.pyc index 832b3c1dbef2e1d6aec9af613900b9f23cf6ac79..3754cc16de0dcbd55c52dff8e8e06d9a7fe6b071 100644 GIT binary patch delta 1448 zcmZ{k&u<$=6vt;~f6dPBI<8~f)@^~3Car1HCMh(*1X@a|go-%ON)_QpWO>{wmg6|H z6Qo45a*SNV7T+CVa7R1GS|>98B#;`Y36Cxb&MyeWP2?0wLA3o ziT0%<-937Y7+-qJpxwZ@(wBkcGL%9_dp;$a%h)&%7$tW}|B6d=xcx09SwP9oi<+C@ z0nOJW(Av;>S^Q0B59r8AI|JKrPTE^e-BT6Vm+d z7?mbn--O1y&mejqqW5Hw`G%+Wu-&YiK(a-MXR{6~17O6D6O4Y3VL$&qy3tc^RJglf z$4}WY*nAiQSRt8AaTyjPhVaUZPjV%j$Cc& z5nE3ibmVUJ*Yt|m8pyG~6SdlU?TgNAua#<5wHy8^dcuC~zeXqQkN)zlQKUCkaj{-o z?!hy5$CO-NAQ4Tl2LbDtT1l^;6b6^@REF`PgmBN>lVJ6R_7XD%y8 z@!yZ3)Ikm$Rk6*mK64%&q&1?eBYLr*S49(ZR2>JosZLlE&ij|pn6p2G7eAPX?kH$) z^tRUfDWtnrqd;{Ay;l4AbP6i*|ESn$F-05pebM{qJ^0xa!9KxKa}bIEtvVH2DJq4P zi%NM1*0nucJ$_0ni+>$afU9`3V_N&?O|15twTi!!;Ki%N??TX zp;iDf-V~b`4pSQMgW|4$LAAIS?v!K-?>_aV~6gdgK zJ1VE1C+dHwo4Do)?}hkURPaXkO`FA+>AZayPd1m)cnjril*8nE9$s8@6{TVSiO{-Kcm-7fas~3P#KsrWb^B8?)yOx4u#4=^6Mq7Scweb2D0m*~x=l*2 Kdu4uxpZ*tNA`<@q delta 1396 zcmb`HO>7%Q6vt;~cV|ClcWsBnX$`JKO4^!y$mOG{Xc{#t>H(=iRU$PcD%Fm zo8RtN)$d=4n^6=n{5^dCVDtFIy?9IPeU06nwieo&WPMxnHXCrwhP>~jVqj?pTQao= z&XA=NIEa0IjA;qhB4*oddjlVAaE3z5hKrRz+aSRO4}1tf?Mu#51);WX3(nTr);XJ{ zVQ;`$-{oxQMCvQ;-sUvOJQ34VKXZJX13t2$+3Sv?=JrR8&1R?5f<`kP@%i8QoLROi z&dBPvw#|lhgGc6p^+DO)yxIBmldUe+(`4i(t?f>8vt_32Z%-Vf^>V>I9`e8{i!om2 zF&-Ws$(;GsuI~L#2<{&tND;u0^&P;4gXrU;fPXBK4x%9-OuIwT_Xvsh5RR{1?Guia zwWap8b3-5|dtcxkoW9gv_d`10s%RI)0xtVjC3OcL;`Bj)M4Wp}JK!QuryaD_AS8Ck z(`f`)iI699gb*TdAkKKkSTAJIVkkT#dmhOaKWDwC$TrB>#yU70RiyJeyUqZw?FQhi z9=7VN&H`YotX5iCONwSqPM4Ar@>wCKgp+qp&HaDNa$G zrZ_|4V~ixEz3o;v@vu`c>R+8tzCJO}SJjPn`Z)|h5Xv`7V z&%#S82TRp=%#Y!lEB^;wy4wYcS#vEq%@<8C+M3(ILYKTdQm$+r374bCuY@@KN-UZ2 j_+%}6Gzb5X?{=nsfx2@P=O`{xESZh?ZKveD>Q(;)^?MA@ diff --git a/crisp/extract/crisp_extract_function.py b/crisp/extract/crisp_extract_function.py index 7be42ce..7c71900 100644 --- a/crisp/extract/crisp_extract_function.py +++ b/crisp/extract/crisp_extract_function.py @@ -18,12 +18,11 @@ def list_website_operators_csv(website_id, client, path, filename): try: output_location = path + filename list_website_operators_json = client.website.list_website_operators(website_id) - list_website_operators_df = general_function.json_normalize_to_df(list_website_operators_json) + list_website_operators_df = general_function.json_normalize_to_df(list_website_operators_json, sep='-') # Datetime add list_website_operators_df['api_called_at'] = date.today() - timedelta(days = 1) - list_website_operators_df.columns = list_website_operators_df.columns.str.replace(".", "-") list_website_operators_df.to_csv(output_location, sep='|', encoding='utf-8-sig', header='true', quotechar= '"', index=False) print("[Download Complete]",filename) except Exception as e: @@ -41,7 +40,7 @@ def list_people_profiles_csv(website_id, client, path, filename): while loop_flag == True: list_people_profiles_json = client.website.list_people_profiles(website_id, page_number) - temp_df = general_function.json_normalize_to_df(list_people_profiles_json) + temp_df = general_function.json_normalize_to_df(list_people_profiles_json, sep='-') temp_df['api_page_number'] = page_number list_people_profiles_df = pd.concat([temp_df,list_people_profiles_df], ignore_index=True) if not list_people_profiles_json: @@ -49,14 +48,13 @@ def list_people_profiles_csv(website_id, client, path, filename): else: page_number += 1 # Timestamp to datetime - list_people_profiles_df['created_at']=(pd.to_datetime(list_people_profiles_df['created_at'],unit='s')) - list_people_profiles_df['updated_at']=(pd.to_datetime(list_people_profiles_df['updated_at'],unit='s')) - list_people_profiles_df['active-last']=(pd.to_datetime(list_people_profiles_df['active-last'],unit='s')) + list_people_profiles_df['created_at']=(pd.to_datetime(list_people_profiles_df['created_at'],unit='ms')) + list_people_profiles_df['updated_at']=(pd.to_datetime(list_people_profiles_df['updated_at'],unit='ms')) + list_people_profiles_df['active-last']=(pd.to_datetime(list_people_profiles_df['active-last'],unit='ms')) # Datetime add list_people_profiles_df['api_called_at'] = date.today() - timedelta(days = 1) - list_people_profiles_df.columns = list_people_profiles_df.columns.str.replace(".", "-") list_people_profiles_df.to_csv(output_location, sep='|', encoding='utf-8-sig', header='true', quotechar= '"', index=False) print("[Download Complete]",filename) @@ -77,14 +75,18 @@ def get_people_data_csv(website_id, client, path, filename, list_people_profiles for people_id in person_ids: get_people_data_json = client.website.get_people_data(website_id, people_id) temp_df = general_function.json_normalize_to_df(get_people_data_json) + + temp_df['people_id'] = people_id get_people_data_df = pd.concat([temp_df,get_people_data_df], ignore_index=True) + # Fix setup error by CX team, eg( exchange-marking, shipping-marking -> should be shipping_marking) + get_people_data_df.columns = get_people_data_df.columns.str.replace("-", "_") + get_people_data_df.columns = get_people_data_df.columns.str.replace(".", "-") + # Datetime add get_people_data_df['api_called_at'] = date.today() - timedelta(days = 1) - get_people_data_df.columns = get_people_data_df.columns.str.replace("-", "_") - get_people_data_df.columns = get_people_data_df.columns.str.replace(".", "-") get_people_data_df.to_csv(output_location, sep='|', encoding='utf-8-sig', header='true', quotechar= '"', index=False) print("[Download Complete]",filename) @@ -105,7 +107,7 @@ def get_people_subscription_status_csv(website_id, client, path, filename, list_ # Get email subscription status for people_id in person_ids: pget_people_subscription_status_df_json = client.website.get_people_subscription_status(website_id, people_id) - temp_df = general_function.json_normalize_to_df(pget_people_subscription_status_df_json) + temp_df = general_function.json_normalize_to_df(pget_people_subscription_status_df_json, sep='-') temp_df['people_id'] = people_id get_people_subscription_status_df = pd.concat([temp_df,get_people_subscription_status_df], ignore_index=True) diff --git a/general_function.py b/general_function.py index f8ccf12..925640f 100644 --- a/general_function.py +++ b/general_function.py @@ -28,9 +28,10 @@ def change_to_current_directory(): os.chdir(script_dir) -def json_normalize_to_df(json): +def json_normalize_to_df(json, sep='.'): + sep_char = sep try: - df = pd.json_normalize(json) + df = pd.json_normalize(json, sep=sep_char) return df except Exception as e: print(e)