mirror of
https://gitlab.com/cief-data/data-management.git
synced 2026-08-19 04:13:57 +00:00
C2M Scrape add
This commit is contained in:
@@ -57,6 +57,7 @@ data-management
|
|||||||
├─ .gitignore # Ignore the file to push into gitlab
|
├─ .gitignore # Ignore the file to push into gitlab
|
||||||
│
|
│
|
||||||
├─ config # Folder store all the config file to use in the code
|
├─ config # Folder store all the config file to use in the code
|
||||||
|
│ ├─ C2M_CONFIG.json # C2M webscraping configuration
|
||||||
│ ├─ CLOUD-STORAGE-ADMIN-CREDENTIAL.json # Get from GCP Service Account cloud-storage-admin-service-account
|
│ ├─ CLOUD-STORAGE-ADMIN-CREDENTIAL.json # Get from GCP Service Account cloud-storage-admin-service-account
|
||||||
│ ├─ CRISP_CONFIG.json # Config for CRISP API key, secret & web-id
|
│ ├─ CRISP_CONFIG.json # Config for CRISP API key, secret & web-id
|
||||||
│ ├─ EXCHANGE_CONFIG.json # Manually config the MySQL ID, Password, etc...
|
│ ├─ EXCHANGE_CONFIG.json # Manually config the MySQL ID, Password, etc...
|
||||||
@@ -64,6 +65,17 @@ data-management
|
|||||||
│ ├─ SHIPPING_CONFIG.json # Manually config the MySQL ID, Password, etc...
|
│ ├─ SHIPPING_CONFIG.json # Manually config the MySQL ID, Password, etc...
|
||||||
│ └─ SHIPPING_EXTRACT_SQL.json # SQL use in MYSQL to extract the data we need to save in temp's csv
|
│ └─ SHIPPING_EXTRACT_SQL.json # SQL use in MYSQL to extract the data we need to save in temp's csv
|
||||||
│
|
│
|
||||||
|
├─ c2m # C2M webscrapping
|
||||||
|
│ ├─ scrape # All script use to scrape the competitor data is in here
|
||||||
|
│ │ ├─ c2m_scrape_app.py # C2M webscrap, call in main.py or seperate run it in your local environment
|
||||||
|
│ │ ├─ c2m_temp_scrape_files # Temp folder to store all the csv or error.png from webscraper (Auto Create)
|
||||||
|
│ │ │ └─ .gitkeep # Keep empty folder
|
||||||
|
│ │ └─ local_run_files # Folder to store all the data or error from webscraper which run at local pc
|
||||||
|
│ │ └─ error # Folder to store error.png from webscraper which run at local pc
|
||||||
|
│ │ └─ .gitkeep # Keep empty folder
|
||||||
|
│ └─ load
|
||||||
|
│ └─ c2m_load_function.py # C2M load code function to import in main.py
|
||||||
|
│
|
||||||
├─ crisp # CRISP-CRM
|
├─ crisp # CRISP-CRM
|
||||||
│ ├─ extract # All crisp extract needed code is in here
|
│ ├─ extract # All crisp extract needed code is in here
|
||||||
│ │ ├─ crisp_extract_function.py # Crisp extract code function to import in main.py
|
│ │ ├─ crisp_extract_function.py # Crisp extract code function to import in main.py
|
||||||
|
|||||||
@@ -0,0 +1,21 @@
|
|||||||
|
from google.cloud import storage
|
||||||
|
|
||||||
|
def login_cloudstorage_credential(credential):
|
||||||
|
try:
|
||||||
|
storage_client = storage.Client.from_service_account_json(credential)
|
||||||
|
return storage_client
|
||||||
|
except Exception as e:
|
||||||
|
print(e)
|
||||||
|
print("[FUNCTION_ERROR]-login_cloudstorage_credential")
|
||||||
|
return False
|
||||||
|
|
||||||
|
def upload_to_bucket(storage_client, blob_name, file_path):
|
||||||
|
try:
|
||||||
|
crisp_bucket = storage_client.get_bucket('exchange_competitor_bucket_raw')
|
||||||
|
blob = crisp_bucket.blob(blob_name)
|
||||||
|
blob.upload_from_filename(file_path)
|
||||||
|
return print("[Upload Complete]",file_path)
|
||||||
|
except Exception as e:
|
||||||
|
print(e)
|
||||||
|
print("[FUNCTION_ERROR]-upload_to_bucket",blob_name)
|
||||||
|
return False
|
||||||
@@ -0,0 +1,152 @@
|
|||||||
|
from playwright.sync_api import Playwright, sync_playwright
|
||||||
|
import re
|
||||||
|
import time
|
||||||
|
import csv
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import pandas as pd
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
|
||||||
|
def run(playwright: Playwright, DATA_PATH, ERROR_PATH, CONFIG) -> None:
|
||||||
|
try:
|
||||||
|
dt = datetime.now()
|
||||||
|
date = dt.strftime('%Y-%m-%d')
|
||||||
|
hour = dt.hour
|
||||||
|
|
||||||
|
RANGE = CONFIG['range']
|
||||||
|
PRODUCT_NAME = CONFIG['product_name']
|
||||||
|
SITE_NAME = CONFIG['site_name']
|
||||||
|
|
||||||
|
browser = playwright.chromium.launch(headless=False)
|
||||||
|
context = browser.new_context()
|
||||||
|
page = context.new_page()
|
||||||
|
|
||||||
|
page.goto(CONFIG["base_url"])
|
||||||
|
page.click("text=选择语言")
|
||||||
|
|
||||||
|
with page.expect_navigation():
|
||||||
|
page.click("text=English (US)")
|
||||||
|
|
||||||
|
page.click("text=Please login to your account or register an account for free. >> a")
|
||||||
|
page.click("input[name=\"login_id\"]")
|
||||||
|
page.fill("input[name=\"login_id\"]", "yamzhenglim@gmail.com")
|
||||||
|
page.click("input[name=\"login_pass\"]")
|
||||||
|
page.fill("input[name=\"login_pass\"]", "Abc_123456")
|
||||||
|
|
||||||
|
with page.expect_navigation():
|
||||||
|
page.click("button:has-text(\"Login\")")
|
||||||
|
|
||||||
|
page.click(":nth-match(:text(\"Request Service\"), 2)")
|
||||||
|
|
||||||
|
time.sleep(2)
|
||||||
|
page.eval_on_selector('.btn-agree', '$(".btn-agree").button( "option", "disabled", true | false )')
|
||||||
|
page.click("text=I Agree")
|
||||||
|
|
||||||
|
for option in CONFIG['options']: # Not Implemented "ALLINK", "GH", "NH", "OB"
|
||||||
|
try:
|
||||||
|
|
||||||
|
option_selector = page.locator("select[name=\"reload[acc_type]\"]")
|
||||||
|
option_selector.select_option(option)
|
||||||
|
|
||||||
|
data = []
|
||||||
|
for i in range (RANGE['start'], RANGE['stop']+1, RANGE['step']):
|
||||||
|
d = {}
|
||||||
|
myr_selector = '#reload_myr_amount'
|
||||||
|
myr_locator = page.locator(myr_selector)
|
||||||
|
myr_locator.clear()
|
||||||
|
myr_locator.fill(f'{i}')
|
||||||
|
|
||||||
|
page.press("text=Transaction Amount", 'Tab')
|
||||||
|
time.sleep(0.5)
|
||||||
|
|
||||||
|
myr = page.evaluate('$("#reload_myr_amount").val()')
|
||||||
|
cny = page.evaluate('$("#reload_cny_amount").val()')
|
||||||
|
|
||||||
|
q_staff = page.query_selector('.lime')
|
||||||
|
if q_staff:
|
||||||
|
staff_status = q_staff.inner_text().strip()
|
||||||
|
else:
|
||||||
|
staff_status = 'Offline'
|
||||||
|
|
||||||
|
orders_selector = page.query_selector('.in-processing')
|
||||||
|
if orders_selector:
|
||||||
|
orders = re.findall('\d+', orders_selector.inner_text())
|
||||||
|
|
||||||
|
svc = page.evaluate('$("#service_charge").val()')
|
||||||
|
total_payable = page.evaluate('$("#total_amount").val()')
|
||||||
|
current_rate = page.query_selector('#ex_rate').text_content()
|
||||||
|
# today_rate = page.query_selector('.today_rate').text_content()
|
||||||
|
# last_rate_update = page.query_selector('span:below(.today_rate)').text_content()
|
||||||
|
last_rate_update = page.query_selector('.text-muted.help-inline').text_content().strip()
|
||||||
|
|
||||||
|
d['product_name'] = PRODUCT_NAME
|
||||||
|
d['site_name'] = SITE_NAME
|
||||||
|
d['service_type'] = option # need confirmation / add to config
|
||||||
|
d['pending_orders'] = int(orders[0])
|
||||||
|
d['CNY'] = cny
|
||||||
|
d['MYR'] = myr
|
||||||
|
d['handeling_fee'] = svc
|
||||||
|
d['payable'] = total_payable
|
||||||
|
d['current_rate'] = current_rate
|
||||||
|
# d['day_rate'] = today_rate
|
||||||
|
d['last_rate_update'] = last_rate_update
|
||||||
|
d['staff_status'] = staff_status
|
||||||
|
d['scrape_date'] = datetime.now().strftime('%Y-%m-%d')
|
||||||
|
d['scrape_hour'] = datetime.now().strftime('%H')
|
||||||
|
|
||||||
|
data.append(d)
|
||||||
|
|
||||||
|
df = pd.DataFrame(data)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
print(e)
|
||||||
|
stamp = datetime.now().strftime('%Y_%m_%d')
|
||||||
|
page.screenshot(path = f"{ERROR_PATH}/ERROR_{stamp}.png")
|
||||||
|
|
||||||
|
df.to_csv(f"{DATA_PATH}/{option}_{date}_{hour}.csv", index=False)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
print(e)
|
||||||
|
stamp = datetime.now().strftime('%Y_%m_%d')
|
||||||
|
page.screenshot(path = f"{ERROR_PATH}/ERROR_{stamp}.png")
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
|
||||||
|
# Define a dictionary
|
||||||
|
conf_dic = {
|
||||||
|
"base_url": "https://www.c2m.my/reload.php",
|
||||||
|
"product_name": "Exchange",
|
||||||
|
"site_name": "c2m",
|
||||||
|
"range":{
|
||||||
|
"start":1,
|
||||||
|
"stop":100001,
|
||||||
|
"step":1000
|
||||||
|
},
|
||||||
|
"options":["TBDF", "ALTR"]
|
||||||
|
}
|
||||||
|
|
||||||
|
# Convert the dictionary to JSON format
|
||||||
|
conf = json.dumps(conf_dic)
|
||||||
|
|
||||||
|
CONFIG = json.loads(conf)
|
||||||
|
|
||||||
|
DATA_PATH = "./c2m/scrape/local_run_files/"
|
||||||
|
ERROR_PATH ="./c2m/scrape/local_run_files/error/"
|
||||||
|
os.makedirs(DATA_PATH, exist_ok=True)
|
||||||
|
|
||||||
|
with sync_playwright() as playwright:
|
||||||
|
run(playwright, DATA_PATH, ERROR_PATH, CONFIG)
|
||||||
|
|
||||||
|
else:
|
||||||
|
def run_app(C2M_FILES_TEMP_STORAGE_PATH, C2M_CONFIG):
|
||||||
|
# dt = datetime.now()
|
||||||
|
# date = dt.strftime('%Y-%m-%d')
|
||||||
|
# hour = dt.hour
|
||||||
|
|
||||||
|
DATA_PATH = C2M_FILES_TEMP_STORAGE_PATH
|
||||||
|
ERROR_PATH = C2M_FILES_TEMP_STORAGE_PATH
|
||||||
|
os.makedirs(DATA_PATH, exist_ok=True)
|
||||||
|
|
||||||
|
with sync_playwright() as playwright:
|
||||||
|
run(playwright, DATA_PATH, ERROR_PATH, C2M_CONFIG)
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
{
|
||||||
|
"base_url": "https://www.c2m.my/reload.php",
|
||||||
|
"product_name": "Exchange",
|
||||||
|
"site_name": "c2m",
|
||||||
|
"range":{
|
||||||
|
"start":1,
|
||||||
|
"stop":100001,
|
||||||
|
"step":1000
|
||||||
|
},
|
||||||
|
"options":["TBDF", "ALTR"]
|
||||||
|
}
|
||||||
@@ -22,6 +22,13 @@ def delete_csv_in_path(path):
|
|||||||
os.unlink(path)
|
os.unlink(path)
|
||||||
print("[Delete Temp Complete]", path)
|
print("[Delete Temp Complete]", path)
|
||||||
|
|
||||||
|
def delete_csv_or_png_in_path(path):
|
||||||
|
for folder, subfolders, files in os.walk(path):
|
||||||
|
for file in files:
|
||||||
|
if file.endswith('.csv') or file.endswith('.png'):
|
||||||
|
path = os.path.join(folder, file)
|
||||||
|
os.unlink(path)
|
||||||
|
print("[Delete Temp Complete]", path)
|
||||||
|
|
||||||
def change_to_current_directory():
|
def change_to_current_directory():
|
||||||
script_dir = os.path.dirname(os.path.realpath(__file__))
|
script_dir = os.path.dirname(os.path.realpath(__file__))
|
||||||
|
|||||||
@@ -8,7 +8,8 @@ import shipping.extract.shipping_extract_function as shipping_extract_func
|
|||||||
import shipping.load.shipping_load_function as shipping_load_func
|
import shipping.load.shipping_load_function as shipping_load_func
|
||||||
import crisp.extract.crisp_extract_function as crisp_extract_func
|
import crisp.extract.crisp_extract_function as crisp_extract_func
|
||||||
import crisp.load.crisp_load_function as crisp_load_func
|
import crisp.load.crisp_load_function as crisp_load_func
|
||||||
|
import c2m.scrape.c2m_scrape_app as c2m_scrape_app
|
||||||
|
import c2m.load.c2m_load_function as c2m_load_func
|
||||||
|
|
||||||
def exchange_extract(EXCHANGE_CSV_TEMP_STORAGE_PATH):
|
def exchange_extract(EXCHANGE_CSV_TEMP_STORAGE_PATH):
|
||||||
#Config file name
|
#Config file name
|
||||||
@@ -127,52 +128,91 @@ def crisp_load_to_cloudstorage(CRISP_CSV_TEMP_STORAGE_PATH):
|
|||||||
file_path = os.path.join(dirpath,file_name)
|
file_path = os.path.join(dirpath,file_name)
|
||||||
crisp_load_func.upload_to_bucket(cirsp_storage_client, file_name, file_path)
|
crisp_load_func.upload_to_bucket(cirsp_storage_client, file_name, file_path)
|
||||||
|
|
||||||
|
def c2m_scrape(C2M_FILES_TEMP_STORAGE_PATH):
|
||||||
|
C2M_CONFIG_FILE = "C2M_CONFIG.json"
|
||||||
|
C2M_FILES_TEMP_STORAGE_PATH = C2M_FILES_TEMP_STORAGE_PATH
|
||||||
|
|
||||||
|
#Read config file
|
||||||
|
C2M_CONFIG = general_function.read_json_file("./config/",C2M_CONFIG_FILE)
|
||||||
|
|
||||||
|
#Run the c2m_scrape_app
|
||||||
|
c2m_scrape_app.run_app(C2M_FILES_TEMP_STORAGE_PATH, C2M_CONFIG)
|
||||||
|
|
||||||
|
def c2m_load_to_cloudstorage(C2M_FILES_TEMP_STORAGE_PATH):
|
||||||
|
#Config file name
|
||||||
|
CLOUDSTORAGE_CREDENTIAL_FILE = "./config/CLOUD_STORAGE_ADMIN_CREDENTIAL.json"
|
||||||
|
|
||||||
|
#Connect to cloudstorage
|
||||||
|
c2m_storage_client = c2m_load_func.login_cloudstorage_credential(CLOUDSTORAGE_CREDENTIAL_FILE)
|
||||||
|
|
||||||
|
#Get the filename and path in temp csv storage
|
||||||
|
temp_file_list = []
|
||||||
|
for (dirpath, dirnames, filenames) in walk(C2M_FILES_TEMP_STORAGE_PATH):
|
||||||
|
temp_file_list.extend(filenames)
|
||||||
|
|
||||||
|
#Loop all the file and upload to cloudstorage
|
||||||
|
for file_name in temp_file_list:
|
||||||
|
file_path = os.path.join(dirpath,file_name)
|
||||||
|
c2m_load_func.upload_to_bucket(c2m_storage_client, file_name, file_path)
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
print("RUNNING MAIN_PY")
|
print("RUNNING MAIN_PY")
|
||||||
|
|
||||||
# Exchange
|
# # Exchange
|
||||||
#Make sure cron in the file directory
|
# #Make sure cron in the file directory
|
||||||
general_function.change_to_current_directory()
|
# general_function.change_to_current_directory()
|
||||||
#Extract Load Exchange Data
|
# #Extract Load Exchange Data
|
||||||
EXCHANGE_CSV_TEMP_STORAGE_PATH = "./exchange/extract/exchange_temp_extract_csv/"
|
# EXCHANGE_CSV_TEMP_STORAGE_PATH = "./exchange/extract/exchange_temp_extract_csv/"
|
||||||
print("STARTING EXCHANGE EXTRACT")
|
# print("STARTING EXCHANGE EXTRACT")
|
||||||
exchange_extract(EXCHANGE_CSV_TEMP_STORAGE_PATH)
|
# exchange_extract(EXCHANGE_CSV_TEMP_STORAGE_PATH)
|
||||||
print("STARTING EXCHANGE LOAD")
|
# print("STARTING EXCHANGE LOAD")
|
||||||
exchange_load_to_cloudstorage(EXCHANGE_CSV_TEMP_STORAGE_PATH)
|
# exchange_load_to_cloudstorage(EXCHANGE_CSV_TEMP_STORAGE_PATH)
|
||||||
#Delete Temp CSV
|
# #Delete Temp CSV
|
||||||
print("DELETING EXCHANGE CSV")
|
# print("DELETING EXCHANGE CSV")
|
||||||
general_function.delete_csv_in_path(EXCHANGE_CSV_TEMP_STORAGE_PATH)
|
# general_function.delete_csv_in_path(EXCHANGE_CSV_TEMP_STORAGE_PATH)
|
||||||
|
|
||||||
# Shipping
|
# # Shipping
|
||||||
|
# #Make sure cron in the file directory
|
||||||
|
# general_function.change_to_current_directory()
|
||||||
|
# #Extract Load SHIPPING Data
|
||||||
|
# SHIPPING_CSV_TEMP_STORAGE_PATH = "./shipping/extract/shipping_temp_extract_csv/"
|
||||||
|
# print("STARTING SHIPPING EXTRACT")
|
||||||
|
# shipping_extract(SHIPPING_CSV_TEMP_STORAGE_PATH)
|
||||||
|
# print("STARTING SHIPPING LOAD")
|
||||||
|
# shipping_load_to_cloudstorage(SHIPPING_CSV_TEMP_STORAGE_PATH)
|
||||||
|
# #Delete Temp CSV
|
||||||
|
# print("DELETING SHIPPING CSV")
|
||||||
|
# general_function.delete_csv_in_path(SHIPPING_CSV_TEMP_STORAGE_PATH)
|
||||||
|
|
||||||
|
# C2M
|
||||||
#Make sure cron in the file directory
|
#Make sure cron in the file directory
|
||||||
general_function.change_to_current_directory()
|
general_function.change_to_current_directory()
|
||||||
#Extract Load SHIPPING Data
|
#Scrape C2M Data
|
||||||
SHIPPING_CSV_TEMP_STORAGE_PATH = "./shipping/extract/shipping_temp_extract_csv/"
|
C2M_FILES_TEMP_STORAGE_PATH = "./c2m/scrape/c2m_temp_scrape_files/"
|
||||||
print("STARTING SHIPPING EXTRACT")
|
print("STARTING C2M SCRAPE")
|
||||||
shipping_extract(SHIPPING_CSV_TEMP_STORAGE_PATH)
|
c2m_scrape(C2M_FILES_TEMP_STORAGE_PATH)
|
||||||
print("STARTING SHIPPING LOAD")
|
print("STARTING C2M LOAD")
|
||||||
shipping_load_to_cloudstorage(SHIPPING_CSV_TEMP_STORAGE_PATH)
|
c2m_load_to_cloudstorage(C2M_FILES_TEMP_STORAGE_PATH)
|
||||||
#Delete Temp CSV
|
#Delete Temp CSV
|
||||||
print("DELETING SHIPPING CSV")
|
print("DELETING SHIPPING CSV")
|
||||||
general_function.delete_csv_in_path(SHIPPING_CSV_TEMP_STORAGE_PATH)
|
general_function.delete_csv_or_png_in_path(C2M_FILES_TEMP_STORAGE_PATH)
|
||||||
|
|
||||||
# CRISP
|
# # CRISP
|
||||||
#Make sure cron in the file directory
|
# #Make sure cron in the file directory
|
||||||
general_function.change_to_current_directory()
|
# general_function.change_to_current_directory()
|
||||||
#Extract Load Exchange Data
|
# #Extract Load Exchange Data
|
||||||
CRISP_CSV_TEMP_STORAGE_PATH = "./crisp/extract/crisp_temp_extract_csv/"
|
# CRISP_CSV_TEMP_STORAGE_PATH = "./crisp/extract/crisp_temp_extract_csv/"
|
||||||
print("STARTING CRISP EXTRACT")
|
# print("STARTING CRISP EXTRACT")
|
||||||
crisp_extract(CRISP_CSV_TEMP_STORAGE_PATH)
|
# crisp_extract(CRISP_CSV_TEMP_STORAGE_PATH)
|
||||||
print("STARTING CRISP LOAD")
|
# print("STARTING CRISP LOAD")
|
||||||
crisp_load_to_cloudstorage(CRISP_CSV_TEMP_STORAGE_PATH)
|
# crisp_load_to_cloudstorage(CRISP_CSV_TEMP_STORAGE_PATH)
|
||||||
#Delete Temp CSV
|
# #Delete Temp CSV
|
||||||
print("DELETING CRISP CSV")
|
# print("DELETING CRISP CSV")
|
||||||
general_function.delete_csv_in_path(CRISP_CSV_TEMP_STORAGE_PATH)
|
# general_function.delete_csv_in_path(CRISP_CSV_TEMP_STORAGE_PATH)
|
||||||
|
|
||||||
print("END MAIN_PY")
|
# print("END MAIN_PY")
|
||||||
print("------------------------------------------------------------------------------------------------")
|
# print("------------------------------------------------------------------------------------------------")
|
||||||
|
|
||||||
if True:
|
if True:
|
||||||
main()
|
main()
|
||||||
+2
-1
@@ -1,4 +1,5 @@
|
|||||||
mysql-connector-python==8.0.29
|
mysql-connector-python==8.0.29
|
||||||
google-cloud-storage==2.5.0
|
google-cloud-storage==2.5.0
|
||||||
crisp-api==1.1.13
|
crisp-api==1.1.13
|
||||||
pandas==1.5.2
|
pandas==1.5.2
|
||||||
|
playwright==1.34.0
|
||||||
Reference in New Issue
Block a user