import time
import threading
import uuid
from datetime import datetime
import os
import psutil
import pandas as pd
import json
idx=0
idx_lock=threading.Lock()
def get_idx():
global idx
with idx_lock:
idx+=1
current_idx=idx
return current_idx
def get_mem():
pid=os.getpid()
proc=psutil.Process(pid)
mem_info=proc.memory_info()
rss=f'{mem_info.rss/1024/1024:.2f} M'
vms=f'{mem_info.vms/1024/1024:.2f} M'
sys_mem=psutil.virtual_memory()
total_mem=f'{sys_mem.total/1024/1024/1024:.2f} G'
avail_mem=f'{sys_mem.available/1024/1024/1024:.2f} G'
used_percent=f'{sys_mem.percent}%'
return f'Memory:PId:{pid},rss:{rss},vms:{vms},total:{total_mem},avail:{avail_mem},used percent:{used_percent}'
class Book:
def __init__(self,id,name,isbn,author,abstract,comment,content,summary,title,topic):
self.id=id
self.name=name
self.isbn=isbn
self.author=author
self.abstract=abstract
self.comment=comment
self.content=content
self.summary=summary
self.title=title
self.topic=topic
def to_dict(self):
return {
'Id':self.id,
'Name':self.name,
'ISBN':self.isbn,
'Author':self.author,
'Abstract':self.abstract,
'Comment':self.comment,
'Content':self.content,
'Summary':self.summary,
'Title':self.title,
'Topic':self.topic
}
def append_data_list_to_json(data_list=[],json_file="",is_first=False,is_last=False):
if data_list is None or len(data_list)<=0:
print(f'data_list is None or empty')
return
json_str=json.dumps(data_list,indent=4).lstrip('[').rstrip(']').strip()
with open(json_file,'a',encoding='utf-8-sig') as json_write_file:
if is_first:
json_write_file.write('[\n')
json_write_file.write(' ')
json_write_file.write(json_str)
elif not is_last:
json_write_file.write(',\n')
json_write_file.write(' ')
json_write_file.write(json_str)
if is_last:
json_write_file.write(',\n')
json_write_file.write(' ')
json_write_file.write(json_str)
json_write_file.write('\n]')
def split_data(start=0,end=0,batch_size=1000000):
if start>=end:
print(f'start {start}>=end {end}')
return
batch_count=(end-start+batch_size+1)//batch_size
batch_arr=range(0,batch_count)
json_file=f'Json_{datetime.now().strftime('%Y%m%d%H%M%S%f')}.json'
data_list=[]
for batch in batch_arr:
start_idx=start+batch*batch_size
end_idx=min(start+batch*batch_size+batch_size,end)
idx_arr=range(start_idx,end_idx)
for a in idx_arr:
bk=Book(a,f'Name_{a}',f'ISBN_{a}_{uuid.uuid4().hex}',f'Author_{a}',f'Abstract_{a}',f'Comment_{a}',f'Content_{a}',f'Summary_{a}',
f'Title_{a}',f'Topic_{a}')
data_list.append(bk.to_dict())
append_data_list_to_json(data_list,json_file,is_first=start_idx==start,is_last=end_idx==end)
print(f'{datetime.now()},batch:{batch+1},start_idx:{start_idx},end_idx:{end_idx},len:{len(data_list)}')
data_list.clear()
def deserialize_json(json_file):
if not os.path.exists(json_file):
print(f'{json_file} does not exist!')
return
try:
with open(json_file,mode='r',encoding='utf-8-sig') as json_read_file:
data_list=json.load(json_read_file)
print(f'{datetime.now()},len:{len(data_list)}')
last_2_list=data_list[-2:]
print(f'len:{len(last_2_list)}\n{last_2_list}')
except json.JSONDecodeError as ex:
print(f'Json Decode Error:{ex}')
except Exception as ex:
print(f'{datetime.now()},{ex}')
if __name__=='__main__':
split_data(3,29000000,500000)
![image]()
![image]()
![image]()
import json
import uuid
from datetime import datetime
import os
import psutil
import threading
import os
import pandas as pd
def get_mem():
pid=os.getpid()
proc=psutil.Process(pid)
mem_info=proc.memory_info()
rss_mem=f'{mem_info.rss/1024/1024:.2f} M'
vms_mem=f'{mem_info.vms/1024/1024:.2f} M'
sys_mem=psutil.virtual_memory()
total_mem=f'{sys_mem.total/1024/1024/1024:.2f} G'
avail_mem=f'{sys_mem.available/1024/1024/1024:.2f} G'
used_percent=f'{sys_mem.percent} %'
return f'memory,PId:{pid},rss mem:{rss_mem},vms mem:{vms_mem},total:{total_mem},avail:{avail_mem},used percent:{used_percent}'
class Book:
def __init__(self,id,name,isbn,author,abstract,comment,content,summary,title,topic):
self.id=id
self.name=name
self.isbn=isbn
self.author=author
self.abstract=abstract
self.comment=comment
self.content=content
self.summary=summary
self.title=title
self.topic=topic
def to_dict(self):
return {
'Id':self.id,
'Name':self.name,
'ISBN':self.isbn,
'Author':self.author,
'Abstract':self.abstract,
'Comment':self.comment,
'Content':self.content,
'Summary':self.summary,
'Title':self.title,
'Topic':self.topic
}
def write_data_list_to_csv(data_list=[],csv_file="",is_header=False):
if data_list is None or len(data_list)<=0:
print(f'{datetime.now()} is null or empty')
return
df=pd.DataFrame(data_list)
df.to_csv(csv_file,mode='a',header=is_header,index=False,encoding='utf-8-sig')
print(f'{datetime.now()},csv file:{csv_file},len:{len(data_list)},{get_mem()}')
def gengerate_data_list(start_idx=0,end_idx=0):
if start_idx>=end_idx:
print(f'Start_idx:{start_idx}>= end_idx:{end_idx}')
return None
data_list=[]
arr=range(start_idx,end_idx)
for a in arr:
bk=Book(a,f'Name_{a}',f'ISBN_{a}_{uuid.uuid4().hex}',
f'Author_{a}',f'Abstract_{a}',f'Comment_{a}',f'Content_{a}',
f'Summary_{a}',f'Title_{a}',f'Topic_{a}')
data_list.append(bk.to_dict())
return data_list
def split_start_end_in_batch(start=0,end=0,batch_size=1000000):
if start>=end:
f'{datetime.now()},start:{start}>=end:{end}'
return
batch_count=(end-start+batch_size+1)//batch_size
print(f'Total batch:{batch_count}')
batch_arr=range(0,batch_count)
data_list=[]
csv_file=f'CSV_{datetime.now().strftime('%Y%m%d%H%M%S%f')}.csv'
for batch in batch_arr:
start_idx=start+batch*batch_size
end_idx=min(start+batch*batch_size+batch_size,end)
data_list=gengerate_data_list(start_idx,end_idx)
if data_list is None or len(data_list)<=0:
print(f'{datetime.now()},data_list is None or empty')
return
print(f'Batch:{batch+1},start_idx:{start_idx},end_idx:{end_idx}')
write_data_list_to_csv(data_list,csv_file,is_header=start==start_idx)
data_list.clear()
if __name__=='__main__':
split_start_end_in_batch(30,100000000,500000)
![image]()
![image]()
![image]()
![image]()
![image]()