( eg Dataset TinyDB ,LevelDB)
场景与目的: 写简单的脚本,执行任务,数据量不是很多, 希望记录历史(像是使用mongodb 一样),和防止重复执行, 同时达到 事件日志的目的。(希望数据可视化,不引入其它数据库,也不需要创建表(sqllite))
0. 常用脚本
1. HistoryUtils.py (class版本)
2. HistoryUtil.py 脚本版
0.DBUtils (class) eg( user_table= DBUtils(table="user") role_table= DBUtils(table="role"))
#!/usr/bin/env python3 # -*- coding: utf-8 -*- # @mail : lshan523@163.com # @Time : 2026/1/15 10:46 # @Author : Sea # @File : DBUtils.py # @Purpose : # @history : # 小文件场景下,使用json文件存储数据记录,每次追加,后续可以查询,修改,删除 # **************************** import os import json import traceback from functools import wraps from typing import Dict, Any, List, Optional import uuid import tempfile import shutil import threading import time import re # 支持不同方法的独立锁 def synchronized_instance_method_with_retry(method_name="add", max_retries=30, timeout=1.0, retry_delay=0.1): """ 为每个方法创建独立的实例锁,支持超时重试 参数: method_name: 方法名,用于生成锁属性名 max_retries: 最大重试次数 timeout: 每次获取锁的超时时间(秒) retry_delay: 重试延迟时间(秒) """ def decorator(func): @wraps(func) def wrapper(self, *args, **kwargs): # 生成锁属性名 if method_name: lock_attr = f"_lock_{method_name}" else: lock_attr = f"_lock_{func.__name__}" # 确保实例有该方法的锁 if not hasattr(self, lock_attr): lock = threading.Lock() setattr(self, lock_attr, lock) lock = getattr(self, lock_attr) retries = 0 last_exception = None while retries <= max_retries: acquired = False try: # 使用退避算法获取锁 acquired = lock.acquire(timeout=timeout) if acquired: return func(self, *args, **kwargs) else: retries += 1 if retries <= max_retries: # 指数退避策略 wait_time = retry_delay * (2 ** retries) # 指数增加 wait_time = min(wait_time, 5.0) # 上限5秒 print(f"[WARNING] {self.__class__.__name__}.{func.__name__}: " f"第 {retries} 次重试,等待 {wait_time:.2f}s...") time.sleep(wait_time) except Exception as e: last_exception = e if acquired: lock.release() raise e finally: if acquired: lock.release() error_msg = f"{self.__class__.__name__}.{func.__name__}: 重试 {max_retries} 次后仍失败" print(f"[ERROR] {error_msg}") raise TimeoutError(error_msg) from last_exception return wrapper return decorator class DBUtils: def __init__(self, table="history",data_dir= "./mydbs/"): #数据记录文件 self.file_path = os.path.join(data_dir,table)+".json" # 创建文件 if not os.path.exists(self.file_path): print("创建文件 "+str(self.file_path)) os.makedirs(os.path.dirname(self.file_path), exist_ok=True) @synchronized_instance_method_with_retry(method_name="save") def save(self, data={},file_path= None): """ 保存数据记录 把文件写入 本地json文件,可以一直追加,后续可以查询 :param file_path: 数据记录文件 eg:./history/history.json :param history: 数据记录 eg:{"name": "张三9", "age": 27} :return: """ if not file_path: file_path = self.file_path if not data: print("data is None") return # history 必须是 dict assert isinstance(data, dict), "data 必须是 dict" data["_id"]=str(uuid.uuid4()).replace("-", "") with open(file_path, 'a', encoding='utf-8') as f: f.write(json.dumps(data, ensure_ascii=False)) f.write("\n") print("保存数据记录") def find_one(self,query: Dict[str, Any]={},file_path: str= None) -> List[Dict[str, Any]]: """ 查询数据记录 :param file_path: 数据记录文件 :param query: {"name": "张三1"} :return: """ print("查询数据记录") file_path = self.file_path result = self.find(query,1) if result: return result[0] return None def find(self,query: Dict[str, Any] = {}, limit: Optional[int] = None, file_path: str = None) -> List[Dict[str, Any]]: """ 查询所有匹配的数据记录 limit 限制, 默认倒序 分块读取,避免内存溢出 Args: file_path: 数据记录文件路径 query: 查询条件字典,如 {"name": "张三1"} limit 限制查询几条结果 eg 10 Returns: 所有匹配的记录列表,按时间倒序排列(最新的在前) """ results = [] if not file_path: file_path = self.file_path try: #使用更简单的分块读取,从文件末尾开始 with open(file_path, 'rb') as f: # 获取文件大小 file_size = os.path.getsize(file_path) # 设置块大小(可根据实际情况调整) chunk_size = 1024 * 1024 # 1MB position = file_size buffer = b"" lines_processed = 0 max_lines_to_process = 10000 # 限制处理的行数,避免无限循环 while position > 0 and lines_processed < max_lines_to_process: # 计算本次读取的起始位置 read_size = min(chunk_size, position) position -= read_size # 移动到读取位置并读取数据 f.seek(position) chunk = f.read(read_size) # 将新读取的数据添加到缓冲区前面 buffer = chunk + buffer # 分割成行处理 while b'\n' in buffer: # 找到第一个完整行 newline_pos = buffer.find(b'\n') if newline_pos == -1: break line_bytes = buffer[:newline_pos] buffer = buffer[newline_pos + 1:] if line_bytes: try: history = json.loads(line_bytes.decode('utf-8')) lines_processed += 1 # 条件匹配 if all(history.get(key) == value for key, value in query.items()): results.append(history) print(f"查询到匹配记录: {history}") if limit and len(results) >= limit: print(f"已达到限制 {limit} 条,停止查询") return results except json.JSONDecodeError: # 可能是跨块的行,将数据放回缓冲区继续处理 buffer = line_bytes + b'\n' + buffer break except UnicodeDecodeError: line_preview = line_bytes[:50] + b"..." if len(line_bytes) > 50 else line_bytes print(f"编码错误,跳过行: {line_preview}") # 如果缓冲区过大,截断一部分 if len(buffer) > chunk_size * 2: buffer = buffer[-chunk_size:] # 处理缓冲区中剩余的数据(文件开头部分) if buffer: lines = buffer.split(b'\n') for line_bytes in lines: if not line_bytes: continue try: history = json.loads(line_bytes.decode('utf-8')) lines_processed += 1 # 条件匹配 if all(history.get(key) == value for key, value in query.items()): results.append(history) print(f"查询到匹配记录: {history}") if limit and len(results) >= limit: print(f"已达到限制 {limit} 条,停止查询") return results except (json.JSONDecodeError, UnicodeDecodeError): continue except Exception as e: traceback.print_exc() print(f"查询失败: {e}") return [] print(f"共查询到 {len(results)} 条匹配记录") return results print(f"共查询到 {len(results)} 条匹配记录") return results @synchronized_instance_method_with_retry(method_name="update") def update(self,query, update, limit=1,file_path=None): """ 更新数据记录 文件: ./history/history.json 内容:{"name": "张三6", "age": 24} \n {"name": "张三7", "age": 25} \n {"name": "张三8", "age": 26} \n 高效更新 - 直接定位并修改 只更新前limit条匹配的记录 Args: file_path: 数据记录文件路径 query: 查询条件 {"name": "张三8", "age": 26} update: 更新内容 {"name": "张三8", "age": 18} limit: 最多更新的条数(默认1) Returns: 实际更新的条数 """ if not file_path: file_path = self.file_path updated_count = 0 if not os.path.exists(file_path): return 0 with open(file_path, 'r+', encoding='utf-8') as f: while updated_count < limit: line_start = f.tell() line = f.readline() if not line: break line = line.rstrip('\n') if not line: continue try: record = json.loads(line) # 检查是否匹配 if all(record.get(k) == v for k, v in query.items()): # 移动到行开始位置 f.seek(line_start) # 更新记录 record.update(update) # 写入更新后的行 updated_line = json.dumps(record, ensure_ascii=False) + '\n' # 读取当前位置到文件末尾 f.readline() # 跳过原始行 remaining = f.read() # 重新定位并写入 f.seek(line_start) f.write(updated_line) if remaining: f.write(remaining) # 截断文件 f.truncate() # 重置文件指针以便继续读取 f.seek(line_start + len(updated_line)) updated_count += 1 print(f"快速更新第{updated_count}条: {record}") if updated_count >= limit: print("~~ 达到更新限制,结束 ~~") break except json.JSONDecodeError: continue print(f"总共更新了 {updated_count} 条记录") return updated_count @synchronized_instance_method_with_retry(method_name="delete") def delete(self, query={}, limit=1, file_path=None): """ 删除数据记录 文件: ./history/history.json 内容:{"name": "张三6", "age": 24} \n {"name": "张三7", "age": 25} \n {"name": "张三8", "age": 26} \n Args: query: 删除条件 eg: {"name": "张三8", "age": 26} file_path: 数据记录文件路径 limit: 删除的条数 Returns: 实际删除 """ if not file_path: file_path = self.file_path if not query and limit == 1: return 0 deleted_count = 0 temp_file_path = None try: # 检查文件是否存在 if not os.path.exists(file_path): print(f"文件 {file_path} 不存在") return 0 # 确保目录存在 os.makedirs(os.path.dirname(file_path), exist_ok=True) # 在同目录下创建临时文件,避免跨磁盘问题 temp_dir = os.path.dirname(os.path.abspath(file_path)) if not temp_dir: # 如果是当前目录 temp_dir = os.path.dirname(os.path.abspath(__file__)) # 创建临时文件 temp_fd, temp_file_path = tempfile.mkstemp(dir=temp_dir, suffix='.tmp') # 读取原文件,过滤记录 with open(file_path, 'r', encoding='utf-8') as src_file, \ open(temp_fd, 'w', encoding='utf-8') as temp_file: for line in src_file: line = line.strip() if not line: temp_file.write('\n') continue try: record = json.loads(line) # 检查是否匹配查询条件 match = True for key, value in query.items(): if key not in record or record[key] != value: match = False break # 如果匹配且尚未达到删除限制,则跳过(删除) if match and (limit == 0 or deleted_count < limit): deleted_count += 1 continue # 否则保留记录 temp_file.write(line + '\n') except json.JSONDecodeError: # 如果不是有效的JSON,保留原样 temp_file.write(line + '\n') # 关闭文件后移动 # 先备份原文件 backup_file = file_path + '.bak' if os.path.exists(file_path): shutil.copy2(file_path, backup_file) # 用临时文件替换原文件 shutil.move(temp_file_path, file_path) # 删除备份文件(可选) if os.path.exists(backup_file): os.remove(backup_file) print(f"删除了 {deleted_count} 条记录") return deleted_count except Exception as e: print(f"删除过程中出现错误: {e}") # 如果发生错误,恢复备份 if 'backup_file' in locals() and os.path.exists(backup_file): print("尝试恢复备份...") try: shutil.move(backup_file, file_path) print("备份已恢复") except: print("恢复备份失败") # 清理临时文件 if temp_file_path and os.path.exists(temp_file_path): try: os.remove(temp_file_path) except: pass return 0 #支持的查询操作符映射到对应的判断函数 QUERY_OPERATORS = { 'eq': lambda item_value, query_value: item_value == query_value, 'ne': lambda item_value, query_value: item_value != query_value, 'lt': lambda item_value, query_value: item_value < query_value, 'lte': lambda item_value, query_value: item_value <= query_value, 'gt': lambda item_value, query_value: item_value > query_value, 'gte': lambda item_value, query_value: item_value >= query_value, 'in': lambda item_value, query_value: item_value in query_value, 'like': lambda item_value, query_value: bool(re.search(str(query_value), str(item_value))), } MY_OP_KEYS = list(QUERY_OPERATORS.keys()) MY_OP_KEYS.append('or') def check_condition(self,item: Dict[str, Any],query: Dict[str, Any]) -> bool: """检查单个记录是否满足所有查询条件""" # 处理其他AND条件(除了or) and_conditions = [] # op 运算符 必须是 query_operators.key, 否则提示 KeyError for op in query.keys(): if op not in self.MY_OP_KEYS: error = f"{op} 不是支持的查询操作符, 应该是,{json.dumps(self.MY_OP_KEYS, ensure_ascii=False)}" print(error) raise Exception(error) # 去掉 or OR条件 后面 单独处理 and_conditions = [op for op in query.keys() if op not in ['or']] if not and_conditions or len(and_conditions)==0: # 避免 只有 or 的条件 return False for op, conditions in query.items(): if op == 'or': # OR条件单独处理 continue if not isinstance(conditions, dict): print(f"条件格式错误: {conditions}") continue for field, value in conditions.items(): if field not in item: return False try: if not self.QUERY_OPERATORS[op](item[field], value): return False except Exception as e: print(f"比较失败: {e}") # 如果比较失败(如类型不匹配),则视为不匹配 return False #都没出错,说明匹配OK return True def check_or_condition(self,item: Dict[str, Any],query: Dict[str, Any]) -> bool: """检查单个记录是否满足所有查询条件""" or_conditions =query.get('or', None) if or_conditions: if not isinstance(or_conditions, dict): print(f"OR条件格式错误: {or_conditions}") return False return self.check_condition(item,or_conditions) else: return False def mutil_find(self,query: Dict[str, Any] = {}, limit: Optional[int] = None, file_path: str = None) -> List[Dict[str, Any]]: """ 查询所有匹配的数据记录 limit 限制,默认倒序 Args: file_path: 数据记录文件路径 query: 查询条件字典,支持以下操作符: - like: 正则匹配,如 {"name": "张.*"} - in: 在列表中,如 {"name": ["张三1", "张三2"]} - or: OR条件组合,支持多个条件 - eq: 等于,如 {"name": "sea"} - ne: 不等于,如 {"name": "sea"} - lt: 小于,如 {"age": 18} - lte: 小于等于,如 {"age": 18} - gt: 大于,如 {"age": 18} - gte: 大于等于,如 {"age": 18} 示例: { "like": {"name": "张"}, # 正则匹配 "in": {"name": ["张三1", "张三2"]}, "or": {"eq": {"ts": "2026-01-19 11:04:46"}}, # OR条件 "eq": {"name": "sea"}, "ne": {"name": "sea"}, "lt": {"age": 50}, "lte": {"age": 18}, "gt": {"age": 50}, "gte": {"age": 18} } limit: 限制查询多少条结果 Returns: 所有匹配的记录列表,按时间倒序排列(最新的在前) """ results = [] if not file_path: file_path = self.file_path # 检查文件是否存在 if not os.path.exists(file_path): print(f"0 条记录,文件不存在: {file_path}") return [] try: #使用更简单的分块读取,从文件末尾开始 with open(file_path, 'rb') as f: # 获取文件大小 file_size = os.path.getsize(file_path) # 设置块大小(可根据实际情况调整) chunk_size = 1024 * 1024 # 1MB position = file_size buffer = b"" lines_processed = 0 max_lines_to_process = 10000 # 限制处理的行数,避免无限循环 while position > 0 and lines_processed < max_lines_to_process: # 计算本次读取的起始位置 read_size = min(chunk_size, position) position -= read_size # 移动到读取位置并读取数据 f.seek(position) chunk = f.read(read_size) # 将新读取的数据添加到缓冲区前面 buffer = chunk + buffer # 分割成行处理 while b'\n' in buffer: # 找到第一个完整行 newline_pos = buffer.find(b'\n') if newline_pos == -1: break line_bytes = buffer[:newline_pos] buffer = buffer[newline_pos + 1:] if line_bytes: try: history = json.loads(line_bytes.decode('utf-8')) lines_processed += 1 # 条件匹配 if self.check_condition(history,query): results.append(history) limit = limit-1 print(f"查询到匹配记录: {history}") if limit <= 0: break else: # 避免数据重复 if self.check_or_condition(history,query): results.append(history) limit = limit-1 print(f"or 查询到匹配记录: {history}") if limit <= 0: break except json.JSONDecodeError: # 可能是跨块的行,将数据放回缓冲区继续处理 buffer = line_bytes + b'\n' + buffer break except UnicodeDecodeError: line_preview = line_bytes[:50] + b"..." if len(line_bytes) > 50 else line_bytes print(f"编码错误,跳过行: {line_preview}") # 如果缓冲区过大,截断一部分 if len(buffer) > chunk_size * 2: buffer = buffer[-chunk_size:] # 处理缓冲区中剩余的数据(文件开头部分) if buffer: lines = buffer.split(b'\n') for line_bytes in lines: if not line_bytes: continue try: history = json.loads(line_bytes.decode('utf-8')) lines_processed += 1 # 条件匹配 if self.check_condition(history,query): results.append(history) limit = limit-1 print(f"查询到匹配记录: {history}") if limit <= 0: break else: # 避免数据重复 if self.check_or_condition(history,query): results.append(history) limit = limit-1 print(f"or 查询到匹配记录: {history}") if limit <= 0: break except (json.JSONDecodeError, UnicodeDecodeError): continue except Exception as e: traceback.print_exc() print(f"查询失败: {e}") return [] print(f"共查询到 {len(results)} 条匹配记录") return results print(f"共查询到 {len(results)} 条匹配记录") return results def test_save_history(): utils = DBUtils(table="history") for i in range(10): utils.save( {"name": "张三"+str(i), "age": i}) utils.find_one( {"name": "张三"+str(i)}) utils.update({"name": "张三"+str(i)}, {"age": 198},5) # utils.delete({"age": 198}, 5) # 测试使用 def test_mutil_find(): utils = DBUtils(table="history1") # utils.save( {"key": "XMN__50", "ts": "2026-01-19 11:04:46"}) # utils.save( {"key": "XMN__50xs", "ts": "2026-01-19 11:04:51"}) # utils.save( {"key": "S__50xs", "ts": "2026-01-19 11:04:53"}) query1 = { "like": {"name": "张三1"}, "eq": {"age": 198}, "or": {"like": {"name": "张三2"}} } results = utils.mutil_find(query=query1, limit=5) # 查询结果处理 for result in results: print(f"找到记录: {result}") if __name__ == '__main__': # utils = HistoryUtils("./history/history1.json") # test_save_history() # rts = utils.find(query={"name": "张三9"}, limit=5) # print(rts) # test_save_history() test_mutil_find()
1. HistoryUtils.py
#!/usr/bin/env python3 # -*- coding: utf-8 -*- # @mail : lshan523@163.com # @Time : 2026/1/15 10:46 # @Author : Sea # @File : HistoryUtil.py # @Purpose : # @history : # 小文件场景下,使用json文件存储历史记录,每次追加,后续可以查询,修改,删除 # **************************** import os import json import traceback from functools import wraps from typing import Dict, Any, List, Optional import uuid import tempfile import shutil import threading import time import re # 支持不同方法的独立锁 def synchronized_instance_method_with_retry(method_name="add", max_retries=30, timeout=1.0, retry_delay=0.1): """ 为每个方法创建独立的实例锁,支持超时重试 参数: method_name: 方法名,用于生成锁属性名 max_retries: 最大重试次数 timeout: 每次获取锁的超时时间(秒) retry_delay: 重试延迟时间(秒) """ def decorator(func): @wraps(func) def wrapper(self, *args, **kwargs): # 生成锁属性名 if method_name: lock_attr = f"_lock_{method_name}" else: lock_attr = f"_lock_{func.__name__}" # 确保实例有该方法的锁 if not hasattr(self, lock_attr): lock = threading.Lock() setattr(self, lock_attr, lock) lock = getattr(self, lock_attr) retries = 0 last_exception = None while retries <= max_retries: acquired = False try: # 使用退避算法获取锁 acquired = lock.acquire(timeout=timeout) if acquired: return func(self, *args, **kwargs) else: retries += 1 if retries <= max_retries: # 指数退避策略 wait_time = retry_delay * (2 ** retries) # 指数增加 wait_time = min(wait_time, 5.0) # 上限5秒 print(f"[WARNING] {self.__class__.__name__}.{func.__name__}: " f"第 {retries} 次重试,等待 {wait_time:.2f}s...") time.sleep(wait_time) except Exception as e: last_exception = e if acquired: lock.release() raise e finally: if acquired: lock.release() error_msg = f"{self.__class__.__name__}.{func.__name__}: 重试 {max_retries} 次后仍失败" print(f"[ERROR] {error_msg}") raise TimeoutError(error_msg) from last_exception return wrapper return decorator class HistoryUtils: def __init__(self, history_file="./history/history.json"): #历史记录文件 self.history_file = history_file @synchronized_instance_method_with_retry(method_name="update") def save(self, history={},history_file= None): """ 保存历史记录 把文件写入 本地json文件,可以一直追加,后续可以查询 :param history_file: 历史记录文件 eg:./history/history.json :param history: 历史记录 eg:{"name": "张三9", "age": 27} :return: """ history_file = self.history_file if not history: print("history is None") return # history 必须是 dict assert isinstance(history, dict), "history 必须是 dict" history["_id"]=str(uuid.uuid4()).replace("-", "") # 创建文件 if not os.path.exists(history_file): print("history.json 不存在 创建文件 "+str(history_file)) os.makedirs(os.path.dirname(history_file), exist_ok=True) with open(history_file, 'a', encoding='utf-8') as f: f.write(json.dumps(history, ensure_ascii=False)) f.write("\n") print("保存历史记录") def find_one(self,query: Dict[str, Any]={},history_file: str= None) -> List[Dict[str, Any]]: """ 查询历史记录 :param history_file: 历史记录文件 :param query: {"name": "张三1"} :return: """ print("查询历史记录") history_file = self.history_file result = self.find(query,1) if result: return result[0] return None def find(self,query: Dict[str, Any] = {}, limit: Optional[int] = None, history_file: str = None) -> List[Dict[str, Any]]: """ 查询所有匹配的历史记录 limit 限制, 默认倒序 分块读取,避免内存溢出 Args: history_file: 历史记录文件路径 query: 查询条件字典,如 {"name": "张三1"} limit 限制查询几条结果 eg 10 Returns: 所有匹配的记录列表,按时间倒序排列(最新的在前) """ results = [] if not history_file: history_file = self.history_file try: #使用更简单的分块读取,从文件末尾开始 with open(history_file, 'rb') as f: # 获取文件大小 file_size = os.path.getsize(history_file) # 设置块大小(可根据实际情况调整) chunk_size = 1024 * 1024 # 1MB position = file_size buffer = b"" lines_processed = 0 max_lines_to_process = 10000 # 限制处理的行数,避免无限循环 while position > 0 and lines_processed < max_lines_to_process: # 计算本次读取的起始位置 read_size = min(chunk_size, position) position -= read_size # 移动到读取位置并读取数据 f.seek(position) chunk = f.read(read_size) # 将新读取的数据添加到缓冲区前面 buffer = chunk + buffer # 分割成行处理 while b'\n' in buffer: # 找到第一个完整行 newline_pos = buffer.find(b'\n') if newline_pos == -1: break line_bytes = buffer[:newline_pos] buffer = buffer[newline_pos + 1:] if line_bytes: try: history = json.loads(line_bytes.decode('utf-8')) lines_processed += 1 # 条件匹配 if all(history.get(key) == value for key, value in query.items()): results.append(history) print(f"查询到匹配记录: {history}") if limit and len(results) >= limit: print(f"已达到限制 {limit} 条,停止查询") return results except json.JSONDecodeError: # 可能是跨块的行,将数据放回缓冲区继续处理 buffer = line_bytes + b'\n' + buffer break except UnicodeDecodeError: line_preview = line_bytes[:50] + b"..." if len(line_bytes) > 50 else line_bytes print(f"编码错误,跳过行: {line_preview}") # 如果缓冲区过大,截断一部分 if len(buffer) > chunk_size * 2: buffer = buffer[-chunk_size:] # 处理缓冲区中剩余的数据(文件开头部分) if buffer: lines = buffer.split(b'\n') for line_bytes in lines: if not line_bytes: continue try: history = json.loads(line_bytes.decode('utf-8')) lines_processed += 1 # 条件匹配 if all(history.get(key) == value for key, value in query.items()): results.append(history) print(f"查询到匹配记录: {history}") if limit and len(results) >= limit: print(f"已达到限制 {limit} 条,停止查询") return results except (json.JSONDecodeError, UnicodeDecodeError): continue except Exception as e: traceback.print_exc() print(f"查询失败: {e}") return [] print(f"共查询到 {len(results)} 条匹配记录") return results print(f"共查询到 {len(results)} 条匹配记录") return results @synchronized_instance_method_with_retry(method_name="update") def update(self,query, update, limit=1,history_file=None): """ 更新历史记录 文件: ./history/history.json 内容:{"name": "张三6", "age": 24} \n {"name": "张三7", "age": 25} \n {"name": "张三8", "age": 26} \n 高效更新 - 直接定位并修改 只更新前limit条匹配的记录 Args: history_file: 历史记录文件路径 query: 查询条件 {"name": "张三8", "age": 26} update: 更新内容 {"name": "张三8", "age": 18} limit: 最多更新的条数(默认1) Returns: 实际更新的条数 """ history_file = self.history_file updated_count = 0 if not os.path.exists(history_file): return 0 with open(history_file, 'r+', encoding='utf-8') as f: while updated_count < limit: line_start = f.tell() line = f.readline() if not line: break line = line.rstrip('\n') if not line: continue try: record = json.loads(line) # 检查是否匹配 if all(record.get(k) == v for k, v in query.items()): # 移动到行开始位置 f.seek(line_start) # 更新记录 record.update(update) # 写入更新后的行 updated_line = json.dumps(record, ensure_ascii=False) + '\n' # 读取当前位置到文件末尾 f.readline() # 跳过原始行 remaining = f.read() # 重新定位并写入 f.seek(line_start) f.write(updated_line) if remaining: f.write(remaining) # 截断文件 f.truncate() # 重置文件指针以便继续读取 f.seek(line_start + len(updated_line)) updated_count += 1 print(f"更新第{updated_count}条: {record}") if updated_count >= limit: print("~~ 达到更新限制,结束 ~~") break except json.JSONDecodeError: continue print(f"总共更新了 {updated_count} 条记录") return updated_count @synchronized_instance_method_with_retry(method_name="update") def delete(self, query={}, limit=1, history_file=None): """ 删除历史记录 文件: ./history/history.json 内容:{"name": "张三6", "age": 24} \n {"name": "张三7", "age": 25} \n {"name": "张三8", "age": 26} \n Args: query: 删除条件 eg: {"name": "张三8", "age": 26} history_file: 历史记录文件路径 limit: 删除的条数 Returns: 实际删除 """ history_file = self.history_file if not query and limit == 1: return 0 deleted_count = 0 temp_file_path = None try: # 检查文件是否存在 if not os.path.exists(history_file): print(f"文件 {history_file} 不存在") return 0 # 确保目录存在 os.makedirs(os.path.dirname(history_file), exist_ok=True) # 在同目录下创建临时文件,避免跨磁盘问题 temp_dir = os.path.dirname(os.path.abspath(history_file)) if not temp_dir: # 如果是当前目录 temp_dir = os.path.dirname(os.path.abspath(__file__)) # 创建临时文件 temp_fd, temp_file_path = tempfile.mkstemp(dir=temp_dir, suffix='.tmp') # 读取原文件,过滤记录 with open(history_file, 'r', encoding='utf-8') as src_file, \ open(temp_fd, 'w', encoding='utf-8') as temp_file: for line in src_file: line = line.strip() if not line: temp_file.write('\n') continue try: record = json.loads(line) # 检查是否匹配查询条件 match = True for key, value in query.items(): if key not in record or record[key] != value: match = False break # 如果匹配且尚未达到删除限制,则跳过(删除) if match and (limit == 0 or deleted_count < limit): deleted_count += 1 continue # 否则保留记录 temp_file.write(line + '\n') except json.JSONDecodeError: # 如果不是有效的JSON,保留原样 temp_file.write(line + '\n') # 关闭文件后移动 # 先备份原文件 backup_file = history_file + '.bak' if os.path.exists(history_file): shutil.copy2(history_file, backup_file) # 用临时文件替换原文件 shutil.move(temp_file_path, history_file) # 删除备份文件(可选) if os.path.exists(backup_file): os.remove(backup_file) print(f"删除了 {deleted_count} 条记录") return deleted_count except Exception as e: print(f"删除过程中出现错误: {e}") # 如果发生错误,恢复备份 if 'backup_file' in locals() and os.path.exists(backup_file): print("尝试恢复备份...") try: shutil.move(backup_file, history_file) print("备份已恢复") except: print("恢复备份失败") # 清理临时文件 if temp_file_path and os.path.exists(temp_file_path): try: os.remove(temp_file_path) except: pass return 0 #支持的查询操作符映射到对应的判断函数 QUERY_OPERATORS = { 'eq': lambda item_value, query_value: item_value == query_value, 'ne': lambda item_value, query_value: item_value != query_value, 'lt': lambda item_value, query_value: item_value < query_value, 'lte': lambda item_value, query_value: item_value <= query_value, 'gt': lambda item_value, query_value: item_value > query_value, 'gte': lambda item_value, query_value: item_value >= query_value, 'in': lambda item_value, query_value: item_value in query_value, 'like': lambda item_value, query_value: bool(re.search(str(query_value), str(item_value))), } MY_OP_KEYS = list(QUERY_OPERATORS.keys()) MY_OP_KEYS.append('or') def check_condition(self,item: Dict[str, Any],query: Dict[str, Any]) -> bool: """检查单个记录是否满足所有查询条件""" # 处理其他AND条件(除了or) and_conditions = [] # op 运算符 必须是 query_operators.key, 否则提示 KeyError for op in query.keys(): if op not in self.MY_OP_KEYS: error = f"{op} 不是支持的查询操作符, 应该是,{json.dumps(self.MY_OP_KEYS, ensure_ascii=False)}" print(error) raise Exception(error) # 去掉 or OR条件 后面 单独处理 and_conditions = [op for op in query.keys() if op not in ['or']] if not and_conditions or len(and_conditions)==0: # 避免 只有 or 的条件 return False for op, conditions in query.items(): if op == 'or': # OR条件单独处理 continue if not isinstance(conditions, dict): print(f"条件格式错误: {conditions}") continue for field, value in conditions.items(): if field not in item: return False try: if not self.QUERY_OPERATORS[op](item[field], value): return False except Exception as e: print(f"比较失败: {e}") # 如果比较失败(如类型不匹配),则视为不匹配 return False #都没出错,说明匹配OK return True def check_or_condition(self,item: Dict[str, Any],query: Dict[str, Any]) -> bool: """检查单个记录是否满足所有查询条件""" or_conditions =query.get('or', None) if or_conditions: if not isinstance(or_conditions, dict): print(f"OR条件格式错误: {or_conditions}") return False return self.check_condition(item,or_conditions) else: return False def mutil_find(self,query: Dict[str, Any] = {}, limit: Optional[int] = None, history_file: str = None) -> List[Dict[str, Any]]: """ 查询所有匹配的历史记录 limit 限制,默认倒序 Args: history_file: 历史记录文件路径 query: 查询条件字典,支持以下操作符: - like: 正则匹配,如 {"name": "张.*"} - in: 在列表中,如 {"name": ["张三1", "张三2"]} - or: OR条件组合,支持多个条件 - eq: 等于,如 {"name": "sea"} - ne: 不等于,如 {"name": "sea"} - lt: 小于,如 {"age": 18} - lte: 小于等于,如 {"age": 18} - gt: 大于,如 {"age": 18} - gte: 大于等于,如 {"age": 18} 示例: { "like": {"name": "张"}, # 正则匹配 "in": {"name": ["张三1", "张三2"]}, "or": {"eq": {"ts": "2026-01-19 11:04:46"}}, # OR条件 "eq": {"name": "sea"}, "ne": {"name": "sea"}, "lt": {"age": 50}, "lte": {"age": 18}, "gt": {"age": 50}, "gte": {"age": 18} } limit: 限制查询多少条结果 Returns: 所有匹配的记录列表,按时间倒序排列(最新的在前) """ results = [] if not history_file: history_file = self.history_file # 检查文件是否存在 if not os.path.exists(history_file): print(f"0条记录,文件不存在: {history_file}") return [] try: #使用更简单的分块读取,从文件末尾开始 with open(history_file, 'rb') as f: # 获取文件大小 file_size = os.path.getsize(history_file) # 设置块大小(可根据实际情况调整) chunk_size = 1024 * 1024 # 1MB position = file_size buffer = b"" lines_processed = 0 max_lines_to_process = 10000 # 限制处理的行数,避免无限循环 while position > 0 and lines_processed < max_lines_to_process: # 计算本次读取的起始位置 read_size = min(chunk_size, position) position -= read_size # 移动到读取位置并读取数据 f.seek(position) chunk = f.read(read_size) # 将新读取的数据添加到缓冲区前面 buffer = chunk + buffer # 分割成行处理 while b'\n' in buffer: # 找到第一个完整行 newline_pos = buffer.find(b'\n') if newline_pos == -1: break line_bytes = buffer[:newline_pos] buffer = buffer[newline_pos + 1:] if line_bytes: try: history = json.loads(line_bytes.decode('utf-8')) lines_processed += 1 # 条件匹配 if self.check_condition(history,query): results.append(history) limit = limit-1 print(f"查询到匹配记录: {history}") if limit <= 0: break else: # 避免数据重复 if self.check_or_condition(history,query): results.append(history) limit = limit-1 print(f"or 查询到匹配记录: {history}") if limit <= 0: break except json.JSONDecodeError: # 可能是跨块的行,将数据放回缓冲区继续处理 buffer = line_bytes + b'\n' + buffer break except UnicodeDecodeError: line_preview = line_bytes[:50] + b"..." if len(line_bytes) > 50 else line_bytes print(f"编码错误,跳过行: {line_preview}") # 如果缓冲区过大,截断一部分 if len(buffer) > chunk_size * 2: buffer = buffer[-chunk_size:] # 处理缓冲区中剩余的数据(文件开头部分) if buffer: lines = buffer.split(b'\n') for line_bytes in lines: if not line_bytes: continue try: history = json.loads(line_bytes.decode('utf-8')) lines_processed += 1 # 条件匹配 if self.check_condition(history,query): results.append(history) limit = limit-1 print(f"查询到匹配记录: {history}") if limit <= 0: break else: # 避免数据重复 if self.check_or_condition(history,query): results.append(history) limit = limit-1 print(f"or 查询到匹配记录: {history}") if limit <= 0: break except (json.JSONDecodeError, UnicodeDecodeError): continue except Exception as e: traceback.print_exc() print(f"查询失败: {e}") return [] print(f"共查询到 {len(results)} 条匹配记录") return results print(f"共查询到 {len(results)} 条匹配记录") return results def test_save_history(): utils = HistoryUtils("./history/history12.json") for i in range(10): utils.save( {"name": "张三"+str(i), "age": i}) utils.find_one( {"name": "张三"+str(i)}) utils.update({"name": "张三"+str(i)}, {"age": 168},5) # utils.delete({"age": 198}, 5) # 测试使用 def test_mutil_find(): utils = HistoryUtils("./history/history12.json") # utils.save( {"key": "XMN__50", "ts": "2026-01-19 11:04:46"}) # utils.save( {"key": "XMN__50xs", "ts": "2026-01-19 11:04:51"}) # utils.save( {"key": "S__50xs", "ts": "2026-01-19 11:04:53"}) query1 = { "like": {"key": "XMN__50"}, "eq": {"ts": "2026-01-19 11:04:46"}, "or": {"eq": {"ts": "2026-01-19 11:04:51"}} } results = utils.mutil_find(query=query1, limit=5) # 查询结果处理 for result in results: print(f"找到记录: {result}") if __name__ == '__main__': # utils = HistoryUtils("./history/history1.json") # test_save_history() # rts = utils.find(query={"name": "张三9"}, limit=5) # print(rts) test_save_history() # test_mutil_find()
2. HistoryUtil.py
#!/usr/bin/env python3 # -*- coding: utf-8 -*- # @mail : lshan523@163.com # @Time : 2026/1/15 10:46 # @Author : Sea # @File : HistoryUtil.py # @Purpose : # @history : # 小文件场景下,使用json文件存储历史记录,每次追加,后续可以查询 # **************************** import os import json import threading import traceback from functools import wraps from typing import Dict, Any, List, Optional import uuid import tempfile import shutil import re import time # 历史记录文件 HISTORY_FILE="./history/history.json" # 全局锁,避免多线程冲突 GLOBAL_LOCK = threading.Lock() def synchronized_instance_method_with_retry(max_retries=10, delay=0.3, backoff=2, timeout=30): """ 高性能版本:针对文件I/O操作优化 - 优先使用简单锁机制 - 只在需要时才启用超时监控 - 减少线程创建频率 """ def decorator(func): @wraps(func) def wrapper(*args, **kwargs): last_exception = None for attempt in range(max_retries + 1): try: with GLOBAL_LOCK: start_time = time.time() # 对于文件操作,通常不需要复杂的超时机制 # 因为文件操作通常是同步的 result = func(*args, **kwargs) # 如果执行时间超过超时阈值,发出警告 execution_time = time.time() - start_time if execution_time > timeout: print(f"Warning: Function took {execution_time:.2f}s (threshold: {timeout}s)") return result except Exception as e: last_exception = e print(f"Attempt {attempt + 1} of {max_retries + 1} failed: {e}") if attempt == max_retries: raise last_exception time.sleep(delay * (backoff ** attempt)) return None return wrapper return decorator def init(): # 创建文件 if not os.path.exists(HISTORY_FILE): print("history.json 不存在 创建文件 "+str(HISTORY_FILE)) os.makedirs(os.path.dirname(HISTORY_FILE), exist_ok=True) # add save or update or delete lock @synchronized_instance_method_with_retry() def save(history={},history_file=HISTORY_FILE): """ 保存历史记录 把文件写入 本地json文件,可以一直追加,后续可以查询 :param history_file: 历史记录文件 eg:./history/history.json :param history: 历史记录 eg:{"name": "张三9", "age": 27} :return: """ if not history: print("history is None") return # history 必须是 dict assert isinstance(history, dict), "history 必须是 dict" history["_id"]=str(uuid.uuid4()).replace("-", "") # 创建文件 if not os.path.exists(history_file): print("history.json 不存在 创建文件 "+str(history_file)) os.makedirs(os.path.dirname(history_file), exist_ok=True) with open(history_file, 'a', encoding='utf-8') as f: f.write(json.dumps(history, ensure_ascii=False)) f.write("\n") print("保存历史记录") def find(query: Dict[str, Any] = {}, limit: Optional[int] = None, history_file: str = HISTORY_FILE) -> List[Dict[str, Any]]: """ 查询所有匹配的历史记录 limit 限制, 默认倒序 分块读取,避免内存溢出 Args: history_file: 历史记录文件路径 query: 查询条件字典,如 {"name": "张三1"} limit 限制查询几条结果 eg 10 Returns: 所有匹配的记录列表,按时间倒序排列(最新的在前) """ results = [] # 检查文件是否存在 if not os.path.exists(history_file): print(f"文件不存在: {history_file}") return [] try: #使用更简单的分块读取,从文件末尾开始 with open(history_file, 'rb') as f: # 获取文件大小 file_size = os.path.getsize(history_file) # 设置块大小(可根据实际情况调整) chunk_size = 1024 * 1024 # 1MB position = file_size buffer = b"" lines_processed = 0 max_lines_to_process = 10000 # 限制处理的行数,避免无限循环 while position > 0 and lines_processed < max_lines_to_process: # 计算本次读取的起始位置 read_size = min(chunk_size, position) position -= read_size # 移动到读取位置并读取数据 f.seek(position) chunk = f.read(read_size) # 将新读取的数据添加到缓冲区前面 buffer = chunk + buffer # 分割成行处理 while b'\n' in buffer: # 找到第一个完整行 newline_pos = buffer.find(b'\n') if newline_pos == -1: break line_bytes = buffer[:newline_pos] buffer = buffer[newline_pos + 1:] if line_bytes: try: history = json.loads(line_bytes.decode('utf-8')) lines_processed += 1 # 条件匹配 if all(history.get(key) == value for key, value in query.items()): results.append(history) print(f"查询到匹配记录: {history}") if limit and len(results) >= limit: print(f"已达到限制 {limit} 条,停止查询") return results except json.JSONDecodeError: # 可能是跨块的行,将数据放回缓冲区继续处理 buffer = line_bytes + b'\n' + buffer break except UnicodeDecodeError: line_preview = line_bytes[:50] + b"..." if len(line_bytes) > 50 else line_bytes print(f"编码错误,跳过行: {line_preview}") # 如果缓冲区过大,截断一部分 if len(buffer) > chunk_size * 2: buffer = buffer[-chunk_size:] # 处理缓冲区中剩余的数据(文件开头部分) if buffer: lines = buffer.split(b'\n') for line_bytes in lines: if not line_bytes: continue try: history = json.loads(line_bytes.decode('utf-8')) lines_processed += 1 # 条件匹配 if all(history.get(key) == value for key, value in query.items()): results.append(history) print(f"查询到匹配记录: {history}") if limit and len(results) >= limit: print(f"已达到限制 {limit} 条,停止查询") return results except (json.JSONDecodeError, UnicodeDecodeError): continue except Exception as e: traceback.print_exc() print(f"查询失败: {e}") return [] print(f"共查询到 {len(results)} 条匹配记录") return results print(f"共查询到 {len(results)} 条匹配记录") return results def find_one(query: Dict[str, Any]={},history_file: str= HISTORY_FILE) -> List[Dict[str, Any]]: """ 查询历史记录 :param history_file: 历史记录文件 :param query: {"name": "张三1"} :return: """ print("查询历史记录", query) result = find(query,1) if result: return result[0] return None @synchronized_instance_method_with_retry() def update(query, update, limit=1,history_file=HISTORY_FILE): """ 更新历史记录 文件: ./history/history.json 内容:{"name": "张三6", "age": 24} \n {"name": "张三7", "age": 25} \n {"name": "张三8", "age": 26} \n 高效更新 - 直接定位并修改 只更新前limit条匹配的记录 Args: history_file: 历史记录文件路径 query: 查询条件 {"name": "张三8", "age": 26} update: 更新内容 {"name": "张三8", "age": 18} limit: 最多更新的条数(默认1) Returns: 实际更新的条数 """ updated_count = 0 if not os.path.exists(history_file): return 0 with open(history_file, 'r+', encoding='utf-8') as f: while updated_count < limit: line_start = f.tell() line = f.readline() if not line: break line = line.rstrip('\n') if not line: continue try: record = json.loads(line) # 检查是否匹配 if all(record.get(k) == v for k, v in query.items()): # 移动到行开始位置 f.seek(line_start) # 更新记录 record.update(update) # 写入更新后的行 updated_line = json.dumps(record, ensure_ascii=False) + '\n' # 读取当前位置到文件末尾 f.readline() # 跳过原始行 remaining = f.read() # 重新定位并写入 f.seek(line_start) f.write(updated_line) if remaining: f.write(remaining) # 截断文件 f.truncate() # 重置文件指针以便继续读取 f.seek(line_start + len(updated_line)) updated_count += 1 print(f"快速更新第{updated_count}条: {record}") if updated_count >= limit: print("~~ 达到更新限制,结束 ~~") break except json.JSONDecodeError: continue print(f"总共更新了 {updated_count} 条记录") return updated_count @synchronized_instance_method_with_retry() def delete(query={}, limit=1, history_file="history/history.json"): """ 删除历史记录 文件: ./history/history.json 内容:{"name": "张三6", "age": 24} \n {"name": "张三7", "age": 25} \n {"name": "张三8", "age": 26} \n Args: query: 删除条件 eg: {"name": "张三8", "age": 26} history_file: 历史记录文件路径 limit: 删除的条数 Returns: 实际删除 """ if not query and limit == 1: return 0 deleted_count = 0 temp_file_path = None try: # 检查文件是否存在 if not os.path.exists(history_file): print(f"文件 {history_file} 不存在") return 0 # 确保目录存在 os.makedirs(os.path.dirname(history_file), exist_ok=True) # 在同目录下创建临时文件,避免跨磁盘问题 temp_dir = os.path.dirname(os.path.abspath(history_file)) if not temp_dir: # 如果是当前目录 temp_dir = os.path.dirname(os.path.abspath(__file__)) # 创建临时文件 temp_fd, temp_file_path = tempfile.mkstemp(dir=temp_dir, suffix='.tmp') # 读取原文件,过滤记录 with open(history_file, 'r', encoding='utf-8') as src_file, \ open(temp_fd, 'w', encoding='utf-8') as temp_file: for line in src_file: line = line.strip() if not line: temp_file.write('\n') continue try: record = json.loads(line) # 检查是否匹配查询条件 match = True for key, value in query.items(): if key not in record or record[key] != value: match = False break # 如果匹配且尚未达到删除限制,则跳过(删除) if match and (limit == 0 or deleted_count < limit): deleted_count += 1 continue # 否则保留记录 temp_file.write(line + '\n') except json.JSONDecodeError: # 如果不是有效的JSON,保留原样 temp_file.write(line + '\n') # 关闭文件后移动 # 先备份原文件 backup_file = history_file + '.bak' if os.path.exists(history_file): shutil.copy2(history_file, backup_file) # 用临时文件替换原文件 shutil.move(temp_file_path, history_file) # 删除备份文件(可选) if os.path.exists(backup_file): os.remove(backup_file) print(f"删除了 {deleted_count} 条记录") return deleted_count except Exception as e: print(f"删除过程中出现错误: {e}") # 如果发生错误,恢复备份 if 'backup_file' in locals() and os.path.exists(backup_file): print("尝试恢复备份...") try: shutil.move(backup_file, history_file) print("备份已恢复") except: print("恢复备份失败") # 清理临时文件 if temp_file_path and os.path.exists(temp_file_path): try: os.remove(temp_file_path) except: pass return 0 #支持的查询操作符映射到对应的判断函数 QUERY_OPERATORS = { 'eq': lambda item_value, query_value: item_value == query_value, 'ne': lambda item_value, query_value: item_value != query_value, 'lt': lambda item_value, query_value: item_value < query_value, 'lte': lambda item_value, query_value: item_value <= query_value, 'gt': lambda item_value, query_value: item_value > query_value, 'gte': lambda item_value, query_value: item_value >= query_value, 'in': lambda item_value, query_value: item_value in query_value, 'like': lambda item_value, query_value: bool(re.search(str(query_value), str(item_value))), } MY_OP_KEYS = list(QUERY_OPERATORS.keys()) MY_OP_KEYS.append('or') def check_condition(item: Dict[str, Any],query: Dict[str, Any]) -> bool: """检查单个记录是否满足所有查询条件""" # 处理其他AND条件(除了or) and_conditions = [] # op 运算符 必须是 query_operators.key, 否则提示 KeyError for op in query.keys(): if op not in MY_OP_KEYS: error = f"{op} 不是支持的查询操作符, 应该是,{json.dumps(MY_OP_KEYS, ensure_ascii=False)}" print(error) raise Exception(error) # 去掉 or OR条件 后面 单独处理 and_conditions = [op for op in query.keys() if op not in ['or']] if not and_conditions or len(and_conditions)==0: # 避免 只有 or 的条件 return False for op, conditions in query.items(): if op == 'or': # OR条件单独处理 continue if not isinstance(conditions, dict): print(f"条件格式错误: {conditions}") continue for field, value in conditions.items(): if field not in item: return False try: if not QUERY_OPERATORS[op](item[field], value): return False except Exception as e: print(f"比较失败: {e}") # 如果比较失败(如类型不匹配),则视为不匹配 return False #都没出错,说明匹配OK return True def check_or_condition(item: Dict[str, Any],query: Dict[str, Any]) -> bool: """检查单个记录是否满足所有查询条件""" or_conditions =query.get('or', None) if or_conditions: if not isinstance(or_conditions, dict): print(f"OR条件格式错误: {or_conditions}") return False return check_condition(item,or_conditions) else: return False def mutil_find(query: Dict[str, Any] = {}, limit: Optional[int] = None, history_file: str = HISTORY_FILE) -> List[Dict[str, Any]]: """ 查询所有匹配的历史记录 limit 限制,默认倒序 Args: history_file: 历史记录文件路径 query: 查询条件字典,支持以下操作符: - like: 正则匹配,如 {"name": "张.*"} - in: 在列表中,如 {"name": ["张三1", "张三2"]} - or: OR条件组合,支持多个条件 - eq: 等于,如 {"name": "sea"} - ne: 不等于,如 {"name": "sea"} - lt: 小于,如 {"age": 18} - lte: 小于等于,如 {"age": 18} - gt: 大于,如 {"age": 18} - gte: 大于等于,如 {"age": 18} 示例: { "like": {"name": "张"}, # 正则匹配 "in": {"name": ["张三1", "张三2"]}, "or": {"eq": {"ts": "2026-01-19 11:04:46"}}, # OR条件 "eq": {"name": "sea"}, "ne": {"name": "sea"}, "lt": {"age": 50}, "lte": {"age": 18}, "gt": {"age": 50}, "gte": {"age": 18} } limit: 限制查询多少条结果 Returns: 所有匹配的记录列表,按时间倒序排列(最新的在前) """ results = [] # 检查文件是否存在 if not os.path.exists(history_file): print(f"文件不存在: {history_file}") return [] try: #使用更简单的分块读取,从文件末尾开始 with open(history_file, 'rb') as f: # 获取文件大小 file_size = os.path.getsize(history_file) # 设置块大小(可根据实际情况调整) chunk_size = 1024 * 1024 # 1MB position = file_size buffer = b"" lines_processed = 0 max_lines_to_process = 10000 # 限制处理的行数,避免无限循环 while position > 0 and lines_processed < max_lines_to_process: # 计算本次读取的起始位置 read_size = min(chunk_size, position) position -= read_size # 移动到读取位置并读取数据 f.seek(position) chunk = f.read(read_size) # 将新读取的数据添加到缓冲区前面 buffer = chunk + buffer # 分割成行处理 while b'\n' in buffer: # 找到第一个完整行 newline_pos = buffer.find(b'\n') if newline_pos == -1: break line_bytes = buffer[:newline_pos] buffer = buffer[newline_pos + 1:] if line_bytes: try: history = json.loads(line_bytes.decode('utf-8')) lines_processed += 1 # 条件匹配 if check_condition(history,query): results.append(history) limit = limit-1 print(f"查询到匹配记录: {history}") if limit <= 0: break else: # 避免数据重复 if check_or_condition(history,query): results.append(history) limit = limit-1 print(f"or 查询到匹配记录: {history}") if limit <= 0: break except json.JSONDecodeError: # 可能是跨块的行,将数据放回缓冲区继续处理 buffer = line_bytes + b'\n' + buffer break except UnicodeDecodeError: line_preview = line_bytes[:50] + b"..." if len(line_bytes) > 50 else line_bytes print(f"编码错误,跳过行: {line_preview}") # 如果缓冲区过大,截断一部分 if len(buffer) > chunk_size * 2: buffer = buffer[-chunk_size:] # 处理缓冲区中剩余的数据(文件开头部分) if buffer: lines = buffer.split(b'\n') for line_bytes in lines: if not line_bytes: continue try: history = json.loads(line_bytes.decode('utf-8')) lines_processed += 1 # 条件匹配 if check_condition(history,query): results.append(history) limit = limit-1 print(f"查询到匹配记录: {history}") if limit <= 0: break else: # 避免数据重复 if check_or_condition(history,query): results.append(history) limit = limit-1 print(f"or 查询到匹配记录: {history}") if limit <= 0: break except (json.JSONDecodeError, UnicodeDecodeError): continue except Exception as e: traceback.print_exc() print(f"查询失败: {e}") return [] print(f"共查询到 {len(results)} 条匹配记录") return results print(f"共查询到 {len(results)} 条匹配记录") return results # 测试使用 def test_mutil_find(): query1 = { "like": {"mark": "哈1"}, "or": {"like": {"mark": "哈2"}}, # "eq": {"name": "张三9"} } results = mutil_find(query=query1, limit=5) # 查询结果处理 for result in results: print(f"找到记录: {result}") def test_save_history(): for i in range(10): save( {"name": "张三"+str(i), "age": i,"mark":"nihao \n 哈哈哈 \n 哈"}) find_one( {"name": "张三"+str(i)}) update({"name": "张三"+str(i)}, {"age": 198},5) if __name__ == '__main__': # test_save_history() # delete({"age": 198}, 5) # find({'name': '张三5'},10) test_mutil_find() # test_save_history()
浙公网安备 33010602011771号