Python write huge data to json and file in batch

import time
import threading
import uuid
from datetime import datetime
import os
import psutil
import pandas as pd
import json

idx=0
idx_lock=threading.Lock()

def get_idx():
    global idx
    with idx_lock:
        idx+=1
        current_idx=idx
        return current_idx

def get_mem():
    pid=os.getpid()
    proc=psutil.Process(pid)
    mem_info=proc.memory_info()
    rss=f'{mem_info.rss/1024/1024:.2f} M'
    vms=f'{mem_info.vms/1024/1024:.2f} M'
    sys_mem=psutil.virtual_memory()
    total_mem=f'{sys_mem.total/1024/1024/1024:.2f} G'
    avail_mem=f'{sys_mem.available/1024/1024/1024:.2f} G'
    used_percent=f'{sys_mem.percent}%'
    return f'Memory:PId:{pid},rss:{rss},vms:{vms},total:{total_mem},avail:{avail_mem},used percent:{used_percent}'


class Book:
    def __init__(self,id,name,isbn,author,abstract,comment,content,summary,title,topic):
        self.id=id
        self.name=name
        self.isbn=isbn
        self.author=author
        self.abstract=abstract
        self.comment=comment
        self.content=content
        self.summary=summary
        self.title=title
        self.topic=topic

    def to_dict(self):
        return {
            'Id':self.id,
            'Name':self.name,
            'ISBN':self.isbn,
            'Author':self.author,
            'Abstract':self.abstract,
            'Comment':self.comment,
            'Content':self.content,
            'Summary':self.summary,
            'Title':self.title,
            'Topic':self.topic
        }

def append_data_list_to_json(data_list=[],json_file="",is_first=False,is_last=False):
    if data_list is None or len(data_list)<=0:
        print(f'data_list is None or empty')
        return

    json_str=json.dumps(data_list,indent=4).lstrip('[').rstrip(']').strip()

    with open(json_file,'a',encoding='utf-8-sig') as json_write_file:
        if is_first:
            json_write_file.write('[\n')
            json_write_file.write('    ')
            json_write_file.write(json_str)

        elif not is_last:
            json_write_file.write(',\n')
            json_write_file.write('    ')
            json_write_file.write(json_str)

        if is_last:
            json_write_file.write(',\n')
            json_write_file.write('    ')
            json_write_file.write(json_str)
            json_write_file.write('\n]')


def split_data(start=0,end=0,batch_size=1000000):
    if start>=end:
        print(f'start {start}>=end {end}')
        return

    batch_count=(end-start+batch_size+1)//batch_size
    batch_arr=range(0,batch_count)
    json_file=f'Json_{datetime.now().strftime('%Y%m%d%H%M%S%f')}.json'
    data_list=[]

    for batch in batch_arr:
        start_idx=start+batch*batch_size
        end_idx=min(start+batch*batch_size+batch_size,end)
        idx_arr=range(start_idx,end_idx)

        for a in idx_arr:
            bk=Book(a,f'Name_{a}',f'ISBN_{a}_{uuid.uuid4().hex}',f'Author_{a}',f'Abstract_{a}',f'Comment_{a}',f'Content_{a}',f'Summary_{a}',
                    f'Title_{a}',f'Topic_{a}')
            data_list.append(bk.to_dict())

        append_data_list_to_json(data_list,json_file,is_first=start_idx==start,is_last=end_idx==end)
        print(f'{datetime.now()},batch:{batch+1},start_idx:{start_idx},end_idx:{end_idx},len:{len(data_list)}')
        data_list.clear()

def deserialize_json(json_file):
    if not os.path.exists(json_file):
        print(f'{json_file} does not exist!')
        return
    try:
        with open(json_file,mode='r',encoding='utf-8-sig') as json_read_file:
            data_list=json.load(json_read_file)
            print(f'{datetime.now()},len:{len(data_list)}')
            last_2_list=data_list[-2:]
            print(f'len:{len(last_2_list)}\n{last_2_list}')

    except json.JSONDecodeError as ex:
        print(f'Json Decode Error:{ex}')
    except Exception as ex:
        print(f'{datetime.now()},{ex}')

if __name__=='__main__':
    split_data(3,29000000,500000)
    

 

 

 

image

 

 

image

 

 

 

image

 

 

import json
import uuid
from datetime import datetime
import os
import psutil
import threading
import os
import pandas as pd

def get_mem():
    pid=os.getpid()
    proc=psutil.Process(pid)
    mem_info=proc.memory_info()
    rss_mem=f'{mem_info.rss/1024/1024:.2f} M'
    vms_mem=f'{mem_info.vms/1024/1024:.2f} M'
    sys_mem=psutil.virtual_memory()
    total_mem=f'{sys_mem.total/1024/1024/1024:.2f} G'
    avail_mem=f'{sys_mem.available/1024/1024/1024:.2f} G'
    used_percent=f'{sys_mem.percent} %'
    return f'memory,PId:{pid},rss mem:{rss_mem},vms mem:{vms_mem},total:{total_mem},avail:{avail_mem},used percent:{used_percent}'

class Book:

    def __init__(self,id,name,isbn,author,abstract,comment,content,summary,title,topic):
        self.id=id
        self.name=name
        self.isbn=isbn
        self.author=author
        self.abstract=abstract
        self.comment=comment
        self.content=content
        self.summary=summary
        self.title=title
        self.topic=topic

    def to_dict(self):
        return {
            'Id':self.id,
            'Name':self.name,
            'ISBN':self.isbn,
            'Author':self.author,
            'Abstract':self.abstract,
            'Comment':self.comment,
            'Content':self.content,
            'Summary':self.summary,
            'Title':self.title,
            'Topic':self.topic
        }

def write_data_list_to_csv(data_list=[],csv_file="",is_header=False):
    if data_list is None or len(data_list)<=0:
        print(f'{datetime.now()} is null or empty')
        return
    df=pd.DataFrame(data_list)
    df.to_csv(csv_file,mode='a',header=is_header,index=False,encoding='utf-8-sig')
    print(f'{datetime.now()},csv file:{csv_file},len:{len(data_list)},{get_mem()}')

def gengerate_data_list(start_idx=0,end_idx=0):
    if start_idx>=end_idx:
        print(f'Start_idx:{start_idx}>= end_idx:{end_idx}')
        return None
    data_list=[]
    arr=range(start_idx,end_idx)
    for a in arr:
        bk=Book(a,f'Name_{a}',f'ISBN_{a}_{uuid.uuid4().hex}',
                f'Author_{a}',f'Abstract_{a}',f'Comment_{a}',f'Content_{a}',
                f'Summary_{a}',f'Title_{a}',f'Topic_{a}')
        data_list.append(bk.to_dict())

    return data_list

def split_start_end_in_batch(start=0,end=0,batch_size=1000000):
    if start>=end:
        f'{datetime.now()},start:{start}>=end:{end}'
        return

    batch_count=(end-start+batch_size+1)//batch_size
    print(f'Total batch:{batch_count}')

    batch_arr=range(0,batch_count)
    data_list=[]
    csv_file=f'CSV_{datetime.now().strftime('%Y%m%d%H%M%S%f')}.csv'
    for batch in batch_arr:
        start_idx=start+batch*batch_size
        end_idx=min(start+batch*batch_size+batch_size,end)
        data_list=gengerate_data_list(start_idx,end_idx)
        if data_list is None or len(data_list)<=0:
            print(f'{datetime.now()},data_list is None or empty')
            return
        print(f'Batch:{batch+1},start_idx:{start_idx},end_idx:{end_idx}')
        write_data_list_to_csv(data_list,csv_file,is_header=start==start_idx)

        data_list.clear()

if __name__=='__main__':
    split_start_end_in_batch(30,100000000,500000)

 

 

image

 

 

 

 

 

 

 

image

 

 

image

 

 

 

 

 

 

 

 

 

image

 

 

 

image

 

posted @ 2026-03-14 22:01  FredGrit  阅读(18)  评论(0)    收藏  举报