( eg Dataset   TinyDB ,LevelDB)

场景与目的:   写简单的脚本,执行任务,数据量不是很多, 希望记录历史(像是使用mongodb 一样),和防止重复执行, 同时达到 事件日志的目的。(希望数据可视化,不引入其它数据库,也不需要创建表(sqllite)) 
  0. 常用脚本

1.  HistoryUtils.py (class版本)
2. HistoryUtil.py 脚本版

0.DBUtils (class) eg(
user_table= DBUtils(table="user") role_table= DBUtils(table="role"))
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
# @mail    : lshan523@163.com
# @Time    : 2026/1/15 10:46
# @Author  : Sea
# @File    : DBUtils.py
# @Purpose :
# @history :
# 小文件场景下,使用json文件存储数据记录,每次追加,后续可以查询,修改,删除
# ****************************
import os
import json
import traceback
from functools import wraps
from typing import Dict, Any, List, Optional
import uuid
import tempfile
import shutil
import threading
import time
import re

# 支持不同方法的独立锁
def synchronized_instance_method_with_retry(method_name="add", max_retries=30,
                                            timeout=1.0, retry_delay=0.1):
    """
    为每个方法创建独立的实例锁,支持超时重试
    参数:
        method_name: 方法名,用于生成锁属性名
        max_retries: 最大重试次数
        timeout: 每次获取锁的超时时间(秒)
        retry_delay: 重试延迟时间(秒)
    """
    def decorator(func):
        @wraps(func)
        def wrapper(self, *args, **kwargs):
            # 生成锁属性名
            if method_name:
                lock_attr = f"_lock_{method_name}"
            else:
                lock_attr = f"_lock_{func.__name__}"
            # 确保实例有该方法的锁
            if not hasattr(self, lock_attr):
                lock = threading.Lock()
                setattr(self, lock_attr, lock)

            lock = getattr(self, lock_attr)
            retries = 0
            last_exception = None
            while retries <= max_retries:
                acquired = False
                try:
                    # 使用退避算法获取锁
                    acquired = lock.acquire(timeout=timeout)
                    if acquired:
                        return func(self, *args, **kwargs)
                    else:
                        retries += 1
                        if retries <= max_retries:
                            # 指数退避策略
                            wait_time = retry_delay * (2 ** retries)  # 指数增加
                            wait_time = min(wait_time, 5.0)  # 上限5秒
                            print(f"[WARNING] {self.__class__.__name__}.{func.__name__}: "
                                  f"第 {retries} 次重试,等待 {wait_time:.2f}s...")
                            time.sleep(wait_time)
                except Exception as e:
                    last_exception = e
                    if acquired:
                        lock.release()
                    raise e
                finally:
                    if acquired:
                        lock.release()

            error_msg = f"{self.__class__.__name__}.{func.__name__}: 重试 {max_retries} 次后仍失败"
            print(f"[ERROR] {error_msg}")
            raise TimeoutError(error_msg) from last_exception
        return wrapper
    return decorator



class DBUtils:
    def __init__(self, table="history",data_dir= "./mydbs/"):
        #数据记录文件
        self.file_path = os.path.join(data_dir,table)+".json"
        # 创建文件
        if not os.path.exists(self.file_path):
            print("创建文件 "+str(self.file_path))
            os.makedirs(os.path.dirname(self.file_path), exist_ok=True)


    @synchronized_instance_method_with_retry(method_name="save")
    def save(self, data={},file_path= None):
        """
        保存数据记录 把文件写入 本地json文件,可以一直追加,后续可以查询
        :param file_path: 数据记录文件 eg:./history/history.json
        :param history: 数据记录 eg:{"name": "张三9", "age": 27}
        :return:
        """
        if not file_path:
            file_path = self.file_path
        if not data:
            print("data is None")
            return
        # history 必须是 dict
        assert isinstance(data, dict), "data 必须是 dict"
        data["_id"]=str(uuid.uuid4()).replace("-", "")
        with open(file_path, 'a', encoding='utf-8') as f:
            f.write(json.dumps(data, ensure_ascii=False))
            f.write("\n")
        print("保存数据记录")



    def find_one(self,query: Dict[str, Any]={},file_path: str= None) -> List[Dict[str, Any]]:
        """
        查询数据记录
        :param file_path: 数据记录文件
        :param query: {"name": "张三1"}
        :return:
        """
        print("查询数据记录")
        file_path = self.file_path
        result = self.find(query,1)
        if result:
            return result[0]
        return None





    def find(self,query: Dict[str, Any] = {}, limit: Optional[int] = None, file_path: str = None) -> List[Dict[str, Any]]:
        """
        查询所有匹配的数据记录 limit 限制, 默认倒序  分块读取,避免内存溢出
        Args:
            file_path: 数据记录文件路径
            query: 查询条件字典,如 {"name": "张三1"}
            limit 限制查询几条结果 eg 10
        Returns:
            所有匹配的记录列表,按时间倒序排列(最新的在前)
        """
        results = []
        if not file_path:
            file_path = self.file_path
        try:
            #使用更简单的分块读取,从文件末尾开始
            with open(file_path, 'rb') as f:
                # 获取文件大小
                file_size = os.path.getsize(file_path)
                # 设置块大小(可根据实际情况调整)
                chunk_size = 1024 * 1024  # 1MB
                position = file_size
                buffer = b""
                lines_processed = 0
                max_lines_to_process = 10000  # 限制处理的行数,避免无限循环
                while position > 0 and lines_processed < max_lines_to_process:
                    # 计算本次读取的起始位置
                    read_size = min(chunk_size, position)
                    position -= read_size
                    # 移动到读取位置并读取数据
                    f.seek(position)
                    chunk = f.read(read_size)
                    # 将新读取的数据添加到缓冲区前面
                    buffer = chunk + buffer
                    # 分割成行处理
                    while b'\n' in buffer:
                        # 找到第一个完整行
                        newline_pos = buffer.find(b'\n')
                        if newline_pos == -1:
                            break
                        line_bytes = buffer[:newline_pos]
                        buffer = buffer[newline_pos + 1:]
                        if line_bytes:
                            try:
                                history = json.loads(line_bytes.decode('utf-8'))
                                lines_processed += 1

                                # 条件匹配
                                if all(history.get(key) == value for key, value in query.items()):
                                    results.append(history)
                                    print(f"查询到匹配记录: {history}")
                                    if limit and len(results) >= limit:
                                        print(f"已达到限制 {limit} 条,停止查询")
                                        return results
                            except json.JSONDecodeError:
                                # 可能是跨块的行,将数据放回缓冲区继续处理
                                buffer = line_bytes + b'\n' + buffer
                                break
                            except UnicodeDecodeError:
                                line_preview = line_bytes[:50] + b"..." if len(line_bytes) > 50 else line_bytes
                                print(f"编码错误,跳过行: {line_preview}")

                    # 如果缓冲区过大,截断一部分
                    if len(buffer) > chunk_size * 2:
                        buffer = buffer[-chunk_size:]
                # 处理缓冲区中剩余的数据(文件开头部分)
                if buffer:
                    lines = buffer.split(b'\n')
                    for line_bytes in lines:
                        if not line_bytes:
                            continue
                        try:
                            history = json.loads(line_bytes.decode('utf-8'))
                            lines_processed += 1

                            # 条件匹配
                            if all(history.get(key) == value for key, value in query.items()):
                                results.append(history)
                                print(f"查询到匹配记录: {history}")
                                if limit and len(results) >= limit:
                                    print(f"已达到限制 {limit} 条,停止查询")
                                    return results
                        except (json.JSONDecodeError, UnicodeDecodeError):
                            continue
        except Exception as e:
            traceback.print_exc()
            print(f"查询失败: {e}")
            return []
        print(f"共查询到 {len(results)} 条匹配记录")
        return results
        print(f"共查询到 {len(results)} 条匹配记录")
        return results


    @synchronized_instance_method_with_retry(method_name="update")
    def update(self,query, update, limit=1,file_path=None):
        """
        更新数据记录  文件: ./history/history.json
        内容:{"name": "张三6", "age": 24}  \n
             {"name": "张三7", "age": 25}  \n
             {"name": "张三8", "age": 26}  \n
        高效更新 - 直接定位并修改
        只更新前limit条匹配的记录
        Args:
            file_path: 数据记录文件路径
            query: 查询条件   {"name": "张三8", "age": 26}
            update: 更新内容   {"name": "张三8", "age": 18}
            limit: 最多更新的条数(默认1)
        Returns:
            实际更新的条数
        """
        if not file_path:
            file_path = self.file_path
        updated_count = 0
        if not os.path.exists(file_path):
            return 0
        with open(file_path, 'r+', encoding='utf-8') as f:
            while updated_count < limit:
                line_start = f.tell()
                line = f.readline()
                if not line:
                    break
                line = line.rstrip('\n')
                if not line:
                    continue
                try:
                    record = json.loads(line)
                    # 检查是否匹配
                    if all(record.get(k) == v for k, v in query.items()):
                        # 移动到行开始位置
                        f.seek(line_start)
                        # 更新记录
                        record.update(update)
                        # 写入更新后的行
                        updated_line = json.dumps(record, ensure_ascii=False) + '\n'
                        # 读取当前位置到文件末尾
                        f.readline()  # 跳过原始行
                        remaining = f.read()
                        # 重新定位并写入
                        f.seek(line_start)
                        f.write(updated_line)
                        if remaining:
                            f.write(remaining)
                        # 截断文件
                        f.truncate()
                        # 重置文件指针以便继续读取
                        f.seek(line_start + len(updated_line))
                        updated_count += 1
                        print(f"快速更新第{updated_count}条: {record}")
                        if updated_count >= limit:
                            print("~~ 达到更新限制,结束 ~~")
                            break
                except json.JSONDecodeError:
                    continue
        print(f"总共更新了 {updated_count} 条记录")
        return updated_count



    @synchronized_instance_method_with_retry(method_name="delete")
    def delete(self, query={}, limit=1, file_path=None):
        """
        删除数据记录
        文件: ./history/history.json
        内容:{"name": "张三6", "age": 24}  \n
             {"name": "张三7", "age": 25}  \n
             {"name": "张三8", "age": 26}  \n
        Args:
            query: 删除条件 eg: {"name": "张三8", "age": 26}
            file_path: 数据记录文件路径
            limit: 删除的条数
        Returns:
            实际删除
        """
        if not file_path:
            file_path = self.file_path
        if not query and limit == 1:
            return 0
        deleted_count = 0
        temp_file_path = None
        try:
            # 检查文件是否存在
            if not os.path.exists(file_path):
                print(f"文件 {file_path} 不存在")
                return 0
            # 确保目录存在
            os.makedirs(os.path.dirname(file_path), exist_ok=True)
            # 在同目录下创建临时文件,避免跨磁盘问题
            temp_dir = os.path.dirname(os.path.abspath(file_path))
            if not temp_dir:  # 如果是当前目录
                temp_dir = os.path.dirname(os.path.abspath(__file__))
            # 创建临时文件
            temp_fd, temp_file_path = tempfile.mkstemp(dir=temp_dir, suffix='.tmp')
            # 读取原文件,过滤记录
            with open(file_path, 'r', encoding='utf-8') as src_file, \
                    open(temp_fd, 'w', encoding='utf-8') as temp_file:
                for line in src_file:
                    line = line.strip()
                    if not line:
                        temp_file.write('\n')
                        continue
                    try:
                        record = json.loads(line)
                        # 检查是否匹配查询条件
                        match = True
                        for key, value in query.items():
                            if key not in record or record[key] != value:
                                match = False
                                break
                        # 如果匹配且尚未达到删除限制,则跳过(删除)
                        if match and (limit == 0 or deleted_count < limit):
                            deleted_count += 1
                            continue
                        # 否则保留记录
                        temp_file.write(line + '\n')
                    except json.JSONDecodeError:
                        # 如果不是有效的JSON,保留原样
                        temp_file.write(line + '\n')
            # 关闭文件后移动
            # 先备份原文件
            backup_file = file_path + '.bak'
            if os.path.exists(file_path):
                shutil.copy2(file_path, backup_file)
            # 用临时文件替换原文件
            shutil.move(temp_file_path, file_path)
            # 删除备份文件(可选)
            if os.path.exists(backup_file):
                os.remove(backup_file)
            print(f"删除了 {deleted_count} 条记录")
            return deleted_count
        except Exception as e:
            print(f"删除过程中出现错误: {e}")
            # 如果发生错误,恢复备份
            if 'backup_file' in locals() and os.path.exists(backup_file):
                print("尝试恢复备份...")
                try:
                    shutil.move(backup_file, file_path)
                    print("备份已恢复")
                except:
                    print("恢复备份失败")
            # 清理临时文件
            if temp_file_path and os.path.exists(temp_file_path):
                try:
                    os.remove(temp_file_path)
                except:
                    pass
            return 0



    #支持的查询操作符映射到对应的判断函数
    QUERY_OPERATORS = {
        'eq': lambda item_value, query_value: item_value == query_value,
        'ne': lambda item_value, query_value: item_value != query_value,
        'lt': lambda item_value, query_value: item_value < query_value,
        'lte': lambda item_value, query_value: item_value <= query_value,
        'gt': lambda item_value, query_value: item_value > query_value,
        'gte': lambda item_value, query_value: item_value >= query_value,
        'in': lambda item_value, query_value: item_value in query_value,
        'like': lambda item_value, query_value: bool(re.search(str(query_value), str(item_value))),
    }
    MY_OP_KEYS = list(QUERY_OPERATORS.keys())
    MY_OP_KEYS.append('or')


    def check_condition(self,item: Dict[str, Any],query: Dict[str, Any]) -> bool:
        """检查单个记录是否满足所有查询条件"""
        # 处理其他AND条件(除了or)
        and_conditions = []
        # op 运算符 必须是 query_operators.key, 否则提示 KeyError
        for op in query.keys():
            if op not in self.MY_OP_KEYS:
                error = f"{op} 不是支持的查询操作符, 应该是,{json.dumps(self.MY_OP_KEYS, ensure_ascii=False)}"
                print(error)
                raise Exception(error)
        # 去掉 or  OR条件 后面 单独处理
        and_conditions = [op for op in query.keys() if op not in ['or']]
        if not and_conditions or  len(and_conditions)==0:
            # 避免 只有 or 的条件
            return False
        for op, conditions in query.items():
            if op == 'or':  # OR条件单独处理
                continue
            if not isinstance(conditions, dict):
                print(f"条件格式错误: {conditions}")
                continue
            for field, value in conditions.items():
                if field not in item:
                    return False
                try:
                    if not self.QUERY_OPERATORS[op](item[field], value):
                        return False
                except Exception as e:
                    print(f"比较失败: {e}")
                    # 如果比较失败(如类型不匹配),则视为不匹配
                    return False
        #都没出错,说明匹配OK
        return True

    def check_or_condition(self,item: Dict[str, Any],query: Dict[str, Any]) -> bool:
        """检查单个记录是否满足所有查询条件"""
        or_conditions =query.get('or', None)
        if or_conditions:
            if not isinstance(or_conditions, dict):
                print(f"OR条件格式错误: {or_conditions}")
                return False
            return self.check_condition(item,or_conditions)
        else:
            return False

    def mutil_find(self,query: Dict[str, Any] = {}, limit: Optional[int] = None,
                   file_path: str = None) -> List[Dict[str, Any]]:
        """
        查询所有匹配的数据记录 limit 限制,默认倒序
        Args:
            file_path: 数据记录文件路径
            query: 查询条件字典,支持以下操作符:
                - like: 正则匹配,如 {"name": "张.*"}
                - in: 在列表中,如 {"name": ["张三1", "张三2"]}
                - or: OR条件组合,支持多个条件
                - eq: 等于,如 {"name": "sea"}
                - ne: 不等于,如 {"name": "sea"}
                - lt: 小于,如 {"age": 18}
                - lte: 小于等于,如 {"age": 18}
                - gt: 大于,如 {"age": 18}
                - gte: 大于等于,如 {"age": 18}
                示例:
                {
                    "like": {"name": "张"},  # 正则匹配
                    "in": {"name": ["张三1", "张三2"]},
                    "or": {"eq": {"ts": "2026-01-19 11:04:46"}},  # OR条件
                    "eq": {"name": "sea"},
                    "ne": {"name": "sea"},
                    "lt": {"age": 50},
                    "lte": {"age": 18},
                    "gt": {"age": 50},
                    "gte": {"age": 18}
                }
            limit: 限制查询多少条结果

        Returns:
            所有匹配的记录列表,按时间倒序排列(最新的在前)
        """
        results = []
        if not file_path:
            file_path = self.file_path
        # 检查文件是否存在
        if not os.path.exists(file_path):
            print(f"0 条记录,文件不存在: {file_path}")
            return []
        try:
            #使用更简单的分块读取,从文件末尾开始
            with open(file_path, 'rb') as f:
                # 获取文件大小
                file_size = os.path.getsize(file_path)
                # 设置块大小(可根据实际情况调整)
                chunk_size = 1024 * 1024  # 1MB
                position = file_size
                buffer = b""
                lines_processed = 0
                max_lines_to_process = 10000  # 限制处理的行数,避免无限循环
                while position > 0 and lines_processed < max_lines_to_process:
                    # 计算本次读取的起始位置
                    read_size = min(chunk_size, position)
                    position -= read_size
                    # 移动到读取位置并读取数据
                    f.seek(position)
                    chunk = f.read(read_size)
                    # 将新读取的数据添加到缓冲区前面
                    buffer = chunk + buffer
                    # 分割成行处理
                    while b'\n' in buffer:
                        # 找到第一个完整行
                        newline_pos = buffer.find(b'\n')
                        if newline_pos == -1:
                            break
                        line_bytes = buffer[:newline_pos]
                        buffer = buffer[newline_pos + 1:]
                        if line_bytes:
                            try:
                                history = json.loads(line_bytes.decode('utf-8'))
                                lines_processed += 1
                                # 条件匹配
                                if self.check_condition(history,query):
                                    results.append(history)
                                    limit = limit-1
                                    print(f"查询到匹配记录: {history}")
                                    if limit <= 0:
                                        break
                                else:
                                    # 避免数据重复
                                    if self.check_or_condition(history,query):
                                        results.append(history)
                                        limit = limit-1
                                        print(f"or 查询到匹配记录: {history}")
                                        if limit <= 0:
                                            break
                            except json.JSONDecodeError:
                                # 可能是跨块的行,将数据放回缓冲区继续处理
                                buffer = line_bytes + b'\n' + buffer
                                break
                            except UnicodeDecodeError:
                                line_preview = line_bytes[:50] + b"..." if len(line_bytes) > 50 else line_bytes
                                print(f"编码错误,跳过行: {line_preview}")

                    # 如果缓冲区过大,截断一部分
                    if len(buffer) > chunk_size * 2:
                        buffer = buffer[-chunk_size:]
                # 处理缓冲区中剩余的数据(文件开头部分)
                if buffer:
                    lines = buffer.split(b'\n')
                    for line_bytes in lines:
                        if not line_bytes:
                            continue
                        try:
                            history = json.loads(line_bytes.decode('utf-8'))
                            lines_processed += 1
                            # 条件匹配
                            if self.check_condition(history,query):
                                results.append(history)
                                limit = limit-1
                                print(f"查询到匹配记录: {history}")
                                if limit <= 0:
                                    break
                            else:
                                # 避免数据重复
                                if self.check_or_condition(history,query):
                                    results.append(history)
                                    limit = limit-1
                                    print(f"or 查询到匹配记录: {history}")
                                    if limit <= 0:
                                        break
                        except (json.JSONDecodeError, UnicodeDecodeError):
                            continue
        except Exception as e:
            traceback.print_exc()
            print(f"查询失败: {e}")
            return []
        print(f"共查询到 {len(results)} 条匹配记录")
        return results
        print(f"共查询到 {len(results)} 条匹配记录")
        return results



def test_save_history():
    utils = DBUtils(table="history")
    for i in range(10):
        utils.save( {"name": "张三"+str(i), "age": i})
        utils.find_one( {"name": "张三"+str(i)})
        utils.update({"name": "张三"+str(i)}, {"age": 198},5)
        # utils.delete({"age": 198}, 5)

# 测试使用
def test_mutil_find():
    utils = DBUtils(table="history1")
    # utils.save( {"key": "XMN__50", "ts": "2026-01-19 11:04:46"})
    # utils.save( {"key": "XMN__50xs", "ts": "2026-01-19 11:04:51"})
    # utils.save( {"key": "S__50xs", "ts": "2026-01-19 11:04:53"})
    query1 = {
        "like": {"name": "张三1"},
        "eq": {"age": 198},
        "or": {"like": {"name": "张三2"}}
    }
    results = utils.mutil_find(query=query1, limit=5)
    # 查询结果处理
    for result in results:
        print(f"找到记录: {result}")


if __name__ == '__main__':
    # utils = HistoryUtils("./history/history1.json")
    # test_save_history()
    # rts = utils.find(query={"name": "张三9"}, limit=5)
    # print(rts)
    # test_save_history()
    test_mutil_find()

 




1.
HistoryUtils.py
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
# @mail    : lshan523@163.com
# @Time    : 2026/1/15 10:46
# @Author  : Sea
# @File    : HistoryUtil.py
# @Purpose :
# @history :
# 小文件场景下,使用json文件存储历史记录,每次追加,后续可以查询,修改,删除
# ****************************
import os
import json
import traceback
from functools import wraps
from typing import Dict, Any, List, Optional
import uuid
import tempfile
import shutil
import threading
import time
import re

# 支持不同方法的独立锁
def synchronized_instance_method_with_retry(method_name="add", max_retries=30,
                                            timeout=1.0, retry_delay=0.1):
    """
    为每个方法创建独立的实例锁,支持超时重试
    参数:
        method_name: 方法名,用于生成锁属性名
        max_retries: 最大重试次数
        timeout: 每次获取锁的超时时间(秒)
        retry_delay: 重试延迟时间(秒)
    """
    def decorator(func):
        @wraps(func)
        def wrapper(self, *args, **kwargs):
            # 生成锁属性名
            if method_name:
                lock_attr = f"_lock_{method_name}"
            else:
                lock_attr = f"_lock_{func.__name__}"
            # 确保实例有该方法的锁
            if not hasattr(self, lock_attr):
                lock = threading.Lock()
                setattr(self, lock_attr, lock)

            lock = getattr(self, lock_attr)
            retries = 0
            last_exception = None
            while retries <= max_retries:
                acquired = False
                try:
                    # 使用退避算法获取锁
                    acquired = lock.acquire(timeout=timeout)
                    if acquired:
                        return func(self, *args, **kwargs)
                    else:
                        retries += 1
                        if retries <= max_retries:
                            # 指数退避策略
                            wait_time = retry_delay * (2 ** retries)  # 指数增加
                            wait_time = min(wait_time, 5.0)  # 上限5秒
                            print(f"[WARNING] {self.__class__.__name__}.{func.__name__}: "
                                  f"第 {retries} 次重试,等待 {wait_time:.2f}s...")
                            time.sleep(wait_time)
                except Exception as e:
                    last_exception = e
                    if acquired:
                        lock.release()
                    raise e
                finally:
                    if acquired:
                        lock.release()

            error_msg = f"{self.__class__.__name__}.{func.__name__}: 重试 {max_retries} 次后仍失败"
            print(f"[ERROR] {error_msg}")
            raise TimeoutError(error_msg) from last_exception
        return wrapper
    return decorator



class HistoryUtils:
    def __init__(self, history_file="./history/history.json"):
        #历史记录文件
        self.history_file = history_file
    @synchronized_instance_method_with_retry(method_name="update")
    def save(self, history={},history_file= None):
        """
        保存历史记录 把文件写入 本地json文件,可以一直追加,后续可以查询
        :param history_file: 历史记录文件 eg:./history/history.json
        :param history: 历史记录 eg:{"name": "张三9", "age": 27}
        :return:
        """
        history_file = self.history_file
        if not history:
            print("history is None")
            return
        # history 必须是 dict
        assert isinstance(history, dict), "history 必须是 dict"
        history["_id"]=str(uuid.uuid4()).replace("-", "")
        # 创建文件
        if not os.path.exists(history_file):
            print("history.json 不存在 创建文件 "+str(history_file))
            os.makedirs(os.path.dirname(history_file), exist_ok=True)
        with open(history_file, 'a', encoding='utf-8') as f:
            f.write(json.dumps(history, ensure_ascii=False))
            f.write("\n")
        print("保存历史记录")



    def find_one(self,query: Dict[str, Any]={},history_file: str= None) -> List[Dict[str, Any]]:
        """
        查询历史记录
        :param history_file: 历史记录文件
        :param query: {"name": "张三1"}
        :return:
        """
        print("查询历史记录")
        history_file = self.history_file
        result = self.find(query,1)
        if result:
            return result[0]
        return None




    def find(self,query: Dict[str, Any] = {}, limit: Optional[int] = None, history_file: str = None) -> List[Dict[str, Any]]:
        """
        查询所有匹配的历史记录 limit 限制, 默认倒序  分块读取,避免内存溢出
        Args:
            history_file: 历史记录文件路径
            query: 查询条件字典,如 {"name": "张三1"}
            limit 限制查询几条结果 eg 10
        Returns:
            所有匹配的记录列表,按时间倒序排列(最新的在前)
        """
        results = []
        if not history_file:
            history_file = self.history_file
        try:
            #使用更简单的分块读取,从文件末尾开始
            with open(history_file, 'rb') as f:
                # 获取文件大小
                file_size = os.path.getsize(history_file)
                # 设置块大小(可根据实际情况调整)
                chunk_size = 1024 * 1024  # 1MB
                position = file_size
                buffer = b""
                lines_processed = 0
                max_lines_to_process = 10000  # 限制处理的行数,避免无限循环
                while position > 0 and lines_processed < max_lines_to_process:
                    # 计算本次读取的起始位置
                    read_size = min(chunk_size, position)
                    position -= read_size
                    # 移动到读取位置并读取数据
                    f.seek(position)
                    chunk = f.read(read_size)
                    # 将新读取的数据添加到缓冲区前面
                    buffer = chunk + buffer
                    # 分割成行处理
                    while b'\n' in buffer:
                        # 找到第一个完整行
                        newline_pos = buffer.find(b'\n')
                        if newline_pos == -1:
                            break
                        line_bytes = buffer[:newline_pos]
                        buffer = buffer[newline_pos + 1:]
                        if line_bytes:
                            try:
                                history = json.loads(line_bytes.decode('utf-8'))
                                lines_processed += 1

                                # 条件匹配
                                if all(history.get(key) == value for key, value in query.items()):
                                    results.append(history)
                                    print(f"查询到匹配记录: {history}")
                                    if limit and len(results) >= limit:
                                        print(f"已达到限制 {limit} 条,停止查询")
                                        return results
                            except json.JSONDecodeError:
                                # 可能是跨块的行,将数据放回缓冲区继续处理
                                buffer = line_bytes + b'\n' + buffer
                                break
                            except UnicodeDecodeError:
                                line_preview = line_bytes[:50] + b"..." if len(line_bytes) > 50 else line_bytes
                                print(f"编码错误,跳过行: {line_preview}")

                    # 如果缓冲区过大,截断一部分
                    if len(buffer) > chunk_size * 2:
                        buffer = buffer[-chunk_size:]
                # 处理缓冲区中剩余的数据(文件开头部分)
                if buffer:
                    lines = buffer.split(b'\n')
                    for line_bytes in lines:
                        if not line_bytes:
                            continue
                        try:
                            history = json.loads(line_bytes.decode('utf-8'))
                            lines_processed += 1

                            # 条件匹配
                            if all(history.get(key) == value for key, value in query.items()):
                                results.append(history)
                                print(f"查询到匹配记录: {history}")
                                if limit and len(results) >= limit:
                                    print(f"已达到限制 {limit} 条,停止查询")
                                    return results
                        except (json.JSONDecodeError, UnicodeDecodeError):
                            continue
        except Exception as e:
            traceback.print_exc()
            print(f"查询失败: {e}")
            return []
        print(f"共查询到 {len(results)} 条匹配记录")
        return results
        print(f"共查询到 {len(results)} 条匹配记录")
        return results


    @synchronized_instance_method_with_retry(method_name="update")
    def update(self,query, update, limit=1,history_file=None):
        """
        更新历史记录  文件: ./history/history.json
        内容:{"name": "张三6", "age": 24}  \n
             {"name": "张三7", "age": 25}  \n
             {"name": "张三8", "age": 26}  \n
        高效更新 - 直接定位并修改
        只更新前limit条匹配的记录
        Args:
            history_file: 历史记录文件路径
            query: 查询条件   {"name": "张三8", "age": 26}
            update: 更新内容   {"name": "张三8", "age": 18}
            limit: 最多更新的条数(默认1)
        Returns:
            实际更新的条数
        """
        history_file = self.history_file
        updated_count = 0
        if not os.path.exists(history_file):
            return 0
        with open(history_file, 'r+', encoding='utf-8') as f:
            while updated_count < limit:
                line_start = f.tell()
                line = f.readline()
                if not line:
                    break
                line = line.rstrip('\n')
                if not line:
                    continue
                try:
                    record = json.loads(line)
                    # 检查是否匹配
                    if all(record.get(k) == v for k, v in query.items()):
                        # 移动到行开始位置
                        f.seek(line_start)
                        # 更新记录
                        record.update(update)
                        # 写入更新后的行
                        updated_line = json.dumps(record, ensure_ascii=False) + '\n'
                        # 读取当前位置到文件末尾
                        f.readline()  # 跳过原始行
                        remaining = f.read()
                        # 重新定位并写入
                        f.seek(line_start)
                        f.write(updated_line)
                        if remaining:
                            f.write(remaining)
                        # 截断文件
                        f.truncate()
                        # 重置文件指针以便继续读取
                        f.seek(line_start + len(updated_line))
                        updated_count += 1
                        print(f"更新第{updated_count}条: {record}")
                        if updated_count >= limit:
                            print("~~ 达到更新限制,结束 ~~")
                            break
                except json.JSONDecodeError:
                    continue
        print(f"总共更新了 {updated_count} 条记录")
        return updated_count



    @synchronized_instance_method_with_retry(method_name="update")
    def delete(self, query={}, limit=1, history_file=None):
        """
        删除历史记录
        文件: ./history/history.json
        内容:{"name": "张三6", "age": 24}  \n
             {"name": "张三7", "age": 25}  \n
             {"name": "张三8", "age": 26}  \n
        Args:
            query: 删除条件 eg: {"name": "张三8", "age": 26}
            history_file: 历史记录文件路径
            limit: 删除的条数
        Returns:
            实际删除
        """
        history_file = self.history_file
        if not query and limit == 1:
            return 0
        deleted_count = 0
        temp_file_path = None
        try:
            # 检查文件是否存在
            if not os.path.exists(history_file):
                print(f"文件 {history_file} 不存在")
                return 0
            # 确保目录存在
            os.makedirs(os.path.dirname(history_file), exist_ok=True)
            # 在同目录下创建临时文件,避免跨磁盘问题
            temp_dir = os.path.dirname(os.path.abspath(history_file))
            if not temp_dir:  # 如果是当前目录
                temp_dir = os.path.dirname(os.path.abspath(__file__))
            # 创建临时文件
            temp_fd, temp_file_path = tempfile.mkstemp(dir=temp_dir, suffix='.tmp')
            # 读取原文件,过滤记录
            with open(history_file, 'r', encoding='utf-8') as src_file, \
                    open(temp_fd, 'w', encoding='utf-8') as temp_file:
                for line in src_file:
                    line = line.strip()
                    if not line:
                        temp_file.write('\n')
                        continue
                    try:
                        record = json.loads(line)
                        # 检查是否匹配查询条件
                        match = True
                        for key, value in query.items():
                            if key not in record or record[key] != value:
                                match = False
                                break
                        # 如果匹配且尚未达到删除限制,则跳过(删除)
                        if match and (limit == 0 or deleted_count < limit):
                            deleted_count += 1
                            continue
                        # 否则保留记录
                        temp_file.write(line + '\n')
                    except json.JSONDecodeError:
                        # 如果不是有效的JSON,保留原样
                        temp_file.write(line + '\n')
            # 关闭文件后移动
            # 先备份原文件
            backup_file = history_file + '.bak'
            if os.path.exists(history_file):
                shutil.copy2(history_file, backup_file)
            # 用临时文件替换原文件
            shutil.move(temp_file_path, history_file)
            # 删除备份文件(可选)
            if os.path.exists(backup_file):
                os.remove(backup_file)
            print(f"删除了 {deleted_count} 条记录")
            return deleted_count
        except Exception as e:
            print(f"删除过程中出现错误: {e}")
            # 如果发生错误,恢复备份
            if 'backup_file' in locals() and os.path.exists(backup_file):
                print("尝试恢复备份...")
                try:
                    shutil.move(backup_file, history_file)
                    print("备份已恢复")
                except:
                    print("恢复备份失败")
            # 清理临时文件
            if temp_file_path and os.path.exists(temp_file_path):
                try:
                    os.remove(temp_file_path)
                except:
                    pass
            return 0



    #支持的查询操作符映射到对应的判断函数
    QUERY_OPERATORS = {
        'eq': lambda item_value, query_value: item_value == query_value,
        'ne': lambda item_value, query_value: item_value != query_value,
        'lt': lambda item_value, query_value: item_value < query_value,
        'lte': lambda item_value, query_value: item_value <= query_value,
        'gt': lambda item_value, query_value: item_value > query_value,
        'gte': lambda item_value, query_value: item_value >= query_value,
        'in': lambda item_value, query_value: item_value in query_value,
        'like': lambda item_value, query_value: bool(re.search(str(query_value), str(item_value))),
    }
    MY_OP_KEYS = list(QUERY_OPERATORS.keys())
    MY_OP_KEYS.append('or')


    def check_condition(self,item: Dict[str, Any],query: Dict[str, Any]) -> bool:
        """检查单个记录是否满足所有查询条件"""
        # 处理其他AND条件(除了or)
        and_conditions = []
        # op 运算符 必须是 query_operators.key, 否则提示 KeyError
        for op in query.keys():
            if op not in self.MY_OP_KEYS:
                error = f"{op} 不是支持的查询操作符, 应该是,{json.dumps(self.MY_OP_KEYS, ensure_ascii=False)}"
                print(error)
                raise Exception(error)
        # 去掉 or  OR条件 后面 单独处理
        and_conditions = [op for op in query.keys() if op not in ['or']]
        if not and_conditions or  len(and_conditions)==0:
            # 避免 只有 or 的条件
            return False
        for op, conditions in query.items():
            if op == 'or':  # OR条件单独处理
                continue
            if not isinstance(conditions, dict):
                print(f"条件格式错误: {conditions}")
                continue
            for field, value in conditions.items():
                if field not in item:
                    return False
                try:
                    if not self.QUERY_OPERATORS[op](item[field], value):
                        return False
                except Exception as e:
                    print(f"比较失败: {e}")
                    # 如果比较失败(如类型不匹配),则视为不匹配
                    return False
        #都没出错,说明匹配OK
        return True

    def check_or_condition(self,item: Dict[str, Any],query: Dict[str, Any]) -> bool:
        """检查单个记录是否满足所有查询条件"""
        or_conditions =query.get('or', None)
        if or_conditions:
            if not isinstance(or_conditions, dict):
                print(f"OR条件格式错误: {or_conditions}")
                return False
            return self.check_condition(item,or_conditions)
        else:
            return False

    def mutil_find(self,query: Dict[str, Any] = {}, limit: Optional[int] = None,
                   history_file: str = None) -> List[Dict[str, Any]]:
        """
        查询所有匹配的历史记录 limit 限制,默认倒序
        Args:
            history_file: 历史记录文件路径
            query: 查询条件字典,支持以下操作符:
                - like: 正则匹配,如 {"name": "张.*"}
                - in: 在列表中,如 {"name": ["张三1", "张三2"]}
                - or: OR条件组合,支持多个条件
                - eq: 等于,如 {"name": "sea"}
                - ne: 不等于,如 {"name": "sea"}
                - lt: 小于,如 {"age": 18}
                - lte: 小于等于,如 {"age": 18}
                - gt: 大于,如 {"age": 18}
                - gte: 大于等于,如 {"age": 18}
                示例:
                {
                    "like": {"name": "张"},  # 正则匹配
                    "in": {"name": ["张三1", "张三2"]},
                    "or": {"eq": {"ts": "2026-01-19 11:04:46"}},  # OR条件
                    "eq": {"name": "sea"},
                    "ne": {"name": "sea"},
                    "lt": {"age": 50},
                    "lte": {"age": 18},
                    "gt": {"age": 50},
                    "gte": {"age": 18}
                }
            limit: 限制查询多少条结果

        Returns:
            所有匹配的记录列表,按时间倒序排列(最新的在前)
        """
        results = []
        if not history_file:
            history_file = self.history_file
        # 检查文件是否存在
        if not os.path.exists(history_file):
            print(f"0条记录,文件不存在: {history_file}")
            return []
        try:
            #使用更简单的分块读取,从文件末尾开始
            with open(history_file, 'rb') as f:
                # 获取文件大小
                file_size = os.path.getsize(history_file)
                # 设置块大小(可根据实际情况调整)
                chunk_size = 1024 * 1024  # 1MB
                position = file_size
                buffer = b""
                lines_processed = 0
                max_lines_to_process = 10000  # 限制处理的行数,避免无限循环
                while position > 0 and lines_processed < max_lines_to_process:
                    # 计算本次读取的起始位置
                    read_size = min(chunk_size, position)
                    position -= read_size
                    # 移动到读取位置并读取数据
                    f.seek(position)
                    chunk = f.read(read_size)
                    # 将新读取的数据添加到缓冲区前面
                    buffer = chunk + buffer
                    # 分割成行处理
                    while b'\n' in buffer:
                        # 找到第一个完整行
                        newline_pos = buffer.find(b'\n')
                        if newline_pos == -1:
                            break
                        line_bytes = buffer[:newline_pos]
                        buffer = buffer[newline_pos + 1:]
                        if line_bytes:
                            try:
                                history = json.loads(line_bytes.decode('utf-8'))
                                lines_processed += 1
                                # 条件匹配
                                if self.check_condition(history,query):
                                    results.append(history)
                                    limit = limit-1
                                    print(f"查询到匹配记录: {history}")
                                    if limit <= 0:
                                        break
                                else:
                                    # 避免数据重复
                                    if self.check_or_condition(history,query):
                                        results.append(history)
                                        limit = limit-1
                                        print(f"or 查询到匹配记录: {history}")
                                        if limit <= 0:
                                            break
                            except json.JSONDecodeError:
                                # 可能是跨块的行,将数据放回缓冲区继续处理
                                buffer = line_bytes + b'\n' + buffer
                                break
                            except UnicodeDecodeError:
                                line_preview = line_bytes[:50] + b"..." if len(line_bytes) > 50 else line_bytes
                                print(f"编码错误,跳过行: {line_preview}")

                    # 如果缓冲区过大,截断一部分
                    if len(buffer) > chunk_size * 2:
                        buffer = buffer[-chunk_size:]
                # 处理缓冲区中剩余的数据(文件开头部分)
                if buffer:
                    lines = buffer.split(b'\n')
                    for line_bytes in lines:
                        if not line_bytes:
                            continue
                        try:
                            history = json.loads(line_bytes.decode('utf-8'))
                            lines_processed += 1
                            # 条件匹配
                            if self.check_condition(history,query):
                                results.append(history)
                                limit = limit-1
                                print(f"查询到匹配记录: {history}")
                                if limit <= 0:
                                    break
                            else:
                                # 避免数据重复
                                if self.check_or_condition(history,query):
                                    results.append(history)
                                    limit = limit-1
                                    print(f"or 查询到匹配记录: {history}")
                                    if limit <= 0:
                                        break
                        except (json.JSONDecodeError, UnicodeDecodeError):
                            continue
        except Exception as e:
            traceback.print_exc()
            print(f"查询失败: {e}")
            return []
        print(f"共查询到 {len(results)} 条匹配记录")
        return results
        print(f"共查询到 {len(results)} 条匹配记录")
        return results



def test_save_history():
    utils = HistoryUtils("./history/history12.json")
    for i in range(10):
        utils.save( {"name": "张三"+str(i), "age": i})
        utils.find_one( {"name": "张三"+str(i)})
        utils.update({"name": "张三"+str(i)}, {"age": 168},5)
        # utils.delete({"age": 198}, 5)

# 测试使用
def test_mutil_find():
    utils = HistoryUtils("./history/history12.json")
    # utils.save( {"key": "XMN__50", "ts": "2026-01-19 11:04:46"})
    # utils.save( {"key": "XMN__50xs", "ts": "2026-01-19 11:04:51"})
    # utils.save( {"key": "S__50xs", "ts": "2026-01-19 11:04:53"})
    query1 = {
        "like": {"key": "XMN__50"},
        "eq": {"ts": "2026-01-19 11:04:46"},
        "or": {"eq": {"ts": "2026-01-19 11:04:51"}}
    }
    results = utils.mutil_find(query=query1, limit=5)
    # 查询结果处理
    for result in results:
        print(f"找到记录: {result}")


if __name__ == '__main__':
    # utils = HistoryUtils("./history/history1.json")
    # test_save_history()
    # rts = utils.find(query={"name": "张三9"}, limit=5)
    # print(rts)
    test_save_history()
    # test_mutil_find()

 

 



2. HistoryUtil.py
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
# @mail    : lshan523@163.com
# @Time    : 2026/1/15 10:46
# @Author  : Sea
# @File    : HistoryUtil.py
# @Purpose :
# @history :
# 小文件场景下,使用json文件存储历史记录,每次追加,后续可以查询
# ****************************
import os
import json
import threading
import traceback
from functools import wraps
from typing import Dict, Any, List, Optional
import uuid
import tempfile
import shutil
import re
import time

# 历史记录文件
HISTORY_FILE="./history/history.json"
# 全局锁,避免多线程冲突
GLOBAL_LOCK = threading.Lock()
def synchronized_instance_method_with_retry(max_retries=10, delay=0.3, backoff=2, timeout=30):
    """
    高性能版本:针对文件I/O操作优化
    - 优先使用简单锁机制
    - 只在需要时才启用超时监控
    - 减少线程创建频率
    """
    def decorator(func):
        @wraps(func)
        def wrapper(*args, **kwargs):
            last_exception = None

            for attempt in range(max_retries + 1):
                try:
                    with GLOBAL_LOCK:
                        start_time = time.time()

                        # 对于文件操作,通常不需要复杂的超时机制
                        # 因为文件操作通常是同步的
                        result = func(*args, **kwargs)

                        # 如果执行时间超过超时阈值,发出警告
                        execution_time = time.time() - start_time
                        if execution_time > timeout:
                            print(f"Warning: Function took {execution_time:.2f}s (threshold: {timeout}s)")

                        return result

                except Exception as e:
                    last_exception = e
                    print(f"Attempt {attempt + 1} of {max_retries + 1} failed: {e}")
                    if attempt == max_retries:
                        raise last_exception
                    time.sleep(delay * (backoff ** attempt))
            return None
        return wrapper
    return decorator


def init():
    # 创建文件
    if not os.path.exists(HISTORY_FILE):
        print("history.json 不存在 创建文件 "+str(HISTORY_FILE))
        os.makedirs(os.path.dirname(HISTORY_FILE), exist_ok=True)

# add save or update or delete  lock


@synchronized_instance_method_with_retry()
def save(history={},history_file=HISTORY_FILE):
    """
    保存历史记录 把文件写入 本地json文件,可以一直追加,后续可以查询
    :param history_file: 历史记录文件 eg:./history/history.json
    :param history: 历史记录 eg:{"name": "张三9", "age": 27}
    :return:
    """
    if not history:
        print("history is None")
        return
    # history 必须是 dict
    assert isinstance(history, dict), "history 必须是 dict"
    history["_id"]=str(uuid.uuid4()).replace("-", "")
    # 创建文件
    if not os.path.exists(history_file):
        print("history.json 不存在 创建文件 "+str(history_file))
        os.makedirs(os.path.dirname(history_file), exist_ok=True)
    with open(history_file, 'a', encoding='utf-8') as f:
        f.write(json.dumps(history, ensure_ascii=False))
        f.write("\n")
    print("保存历史记录")





def find(query: Dict[str, Any] = {}, limit: Optional[int] = None, history_file: str = HISTORY_FILE) -> List[Dict[str, Any]]:
    """
    查询所有匹配的历史记录 limit 限制, 默认倒序  分块读取,避免内存溢出
    Args:
        history_file: 历史记录文件路径
        query: 查询条件字典,如 {"name": "张三1"}
        limit 限制查询几条结果 eg 10
    Returns:
        所有匹配的记录列表,按时间倒序排列(最新的在前)
    """
    results = []

    # 检查文件是否存在
    if not os.path.exists(history_file):
        print(f"文件不存在: {history_file}")
        return []

    try:
        #使用更简单的分块读取,从文件末尾开始
        with open(history_file, 'rb') as f:
            # 获取文件大小
            file_size = os.path.getsize(history_file)
            # 设置块大小(可根据实际情况调整)
            chunk_size = 1024 * 1024  # 1MB
            position = file_size
            buffer = b""
            lines_processed = 0
            max_lines_to_process = 10000  # 限制处理的行数,避免无限循环
            while position > 0 and lines_processed < max_lines_to_process:
                # 计算本次读取的起始位置
                read_size = min(chunk_size, position)
                position -= read_size
                # 移动到读取位置并读取数据
                f.seek(position)
                chunk = f.read(read_size)
                # 将新读取的数据添加到缓冲区前面
                buffer = chunk + buffer
                # 分割成行处理
                while b'\n' in buffer:
                    # 找到第一个完整行
                    newline_pos = buffer.find(b'\n')
                    if newline_pos == -1:
                        break
                    line_bytes = buffer[:newline_pos]
                    buffer = buffer[newline_pos + 1:]
                    if line_bytes:
                        try:
                            history = json.loads(line_bytes.decode('utf-8'))
                            lines_processed += 1

                            # 条件匹配
                            if all(history.get(key) == value for key, value in query.items()):
                                results.append(history)
                                print(f"查询到匹配记录: {history}")
                                if limit and len(results) >= limit:
                                    print(f"已达到限制 {limit} 条,停止查询")
                                    return results
                        except json.JSONDecodeError:
                            # 可能是跨块的行,将数据放回缓冲区继续处理
                            buffer = line_bytes + b'\n' + buffer
                            break
                        except UnicodeDecodeError:
                            line_preview = line_bytes[:50] + b"..." if len(line_bytes) > 50 else line_bytes
                            print(f"编码错误,跳过行: {line_preview}")

                # 如果缓冲区过大,截断一部分
                if len(buffer) > chunk_size * 2:
                    buffer = buffer[-chunk_size:]
            # 处理缓冲区中剩余的数据(文件开头部分)
            if buffer:
                lines = buffer.split(b'\n')
                for line_bytes in lines:
                    if not line_bytes:
                        continue
                    try:
                        history = json.loads(line_bytes.decode('utf-8'))
                        lines_processed += 1

                        # 条件匹配
                        if all(history.get(key) == value for key, value in query.items()):
                            results.append(history)
                            print(f"查询到匹配记录: {history}")
                            if limit and len(results) >= limit:
                                print(f"已达到限制 {limit} 条,停止查询")
                                return results
                    except (json.JSONDecodeError, UnicodeDecodeError):
                        continue
    except Exception as e:
        traceback.print_exc()
        print(f"查询失败: {e}")
        return []
    print(f"共查询到 {len(results)} 条匹配记录")
    return results
    print(f"共查询到 {len(results)} 条匹配记录")
    return results





def find_one(query: Dict[str, Any]={},history_file: str= HISTORY_FILE) -> List[Dict[str, Any]]:
    """
    查询历史记录
    :param history_file: 历史记录文件
    :param query: {"name": "张三1"}
    :return:
    """
    print("查询历史记录", query)
    result = find(query,1)
    if result:
        return result[0]
    return None

@synchronized_instance_method_with_retry()
def update(query, update, limit=1,history_file=HISTORY_FILE):
    """
    更新历史记录  文件: ./history/history.json
    内容:{"name": "张三6", "age": 24}  \n
         {"name": "张三7", "age": 25}  \n
         {"name": "张三8", "age": 26}  \n
    高效更新 - 直接定位并修改
    只更新前limit条匹配的记录
    Args:
        history_file: 历史记录文件路径
        query: 查询条件   {"name": "张三8", "age": 26}
        update: 更新内容   {"name": "张三8", "age": 18}
        limit: 最多更新的条数(默认1)
    Returns:
        实际更新的条数
    """
    updated_count = 0
    if not os.path.exists(history_file):
        return 0
    with open(history_file, 'r+', encoding='utf-8') as f:
        while updated_count < limit:
            line_start = f.tell()
            line = f.readline()
            if not line:
                break
            line = line.rstrip('\n')
            if not line:
                continue
            try:
                record = json.loads(line)
                # 检查是否匹配
                if all(record.get(k) == v for k, v in query.items()):
                    # 移动到行开始位置
                    f.seek(line_start)
                    # 更新记录
                    record.update(update)
                    # 写入更新后的行
                    updated_line = json.dumps(record, ensure_ascii=False) + '\n'
                    # 读取当前位置到文件末尾
                    f.readline()  # 跳过原始行
                    remaining = f.read()
                    # 重新定位并写入
                    f.seek(line_start)
                    f.write(updated_line)
                    if remaining:
                        f.write(remaining)
                    # 截断文件
                    f.truncate()
                    # 重置文件指针以便继续读取
                    f.seek(line_start + len(updated_line))
                    updated_count += 1
                    print(f"快速更新第{updated_count}条: {record}")
                    if updated_count >= limit:
                        print("~~ 达到更新限制,结束 ~~")
                        break
            except json.JSONDecodeError:
                continue
    print(f"总共更新了 {updated_count} 条记录")
    return updated_count

@synchronized_instance_method_with_retry()
def delete(query={}, limit=1, history_file="history/history.json"):
    """
    删除历史记录
    文件: ./history/history.json
    内容:{"name": "张三6", "age": 24}  \n
         {"name": "张三7", "age": 25}  \n
         {"name": "张三8", "age": 26}  \n
    Args:
        query: 删除条件 eg: {"name": "张三8", "age": 26}
        history_file: 历史记录文件路径
        limit: 删除的条数
    Returns:
        实际删除
    """
    if not query and limit == 1:
        return 0
    deleted_count = 0
    temp_file_path = None
    try:
        # 检查文件是否存在
        if not os.path.exists(history_file):
            print(f"文件 {history_file} 不存在")
            return 0
        # 确保目录存在
        os.makedirs(os.path.dirname(history_file), exist_ok=True)
        # 在同目录下创建临时文件,避免跨磁盘问题
        temp_dir = os.path.dirname(os.path.abspath(history_file))
        if not temp_dir:  # 如果是当前目录
            temp_dir = os.path.dirname(os.path.abspath(__file__))
        # 创建临时文件
        temp_fd, temp_file_path = tempfile.mkstemp(dir=temp_dir, suffix='.tmp')
        # 读取原文件,过滤记录
        with open(history_file, 'r', encoding='utf-8') as src_file, \
                open(temp_fd, 'w', encoding='utf-8') as temp_file:
            for line in src_file:
                line = line.strip()
                if not line:
                    temp_file.write('\n')
                    continue
                try:
                    record = json.loads(line)
                    # 检查是否匹配查询条件
                    match = True
                    for key, value in query.items():
                        if key not in record or record[key] != value:
                            match = False
                            break
                    # 如果匹配且尚未达到删除限制,则跳过(删除)
                    if match and (limit == 0 or deleted_count < limit):
                        deleted_count += 1
                        continue
                    # 否则保留记录
                    temp_file.write(line + '\n')
                except json.JSONDecodeError:
                    # 如果不是有效的JSON,保留原样
                    temp_file.write(line + '\n')
        # 关闭文件后移动
        # 先备份原文件
        backup_file = history_file + '.bak'
        if os.path.exists(history_file):
            shutil.copy2(history_file, backup_file)
        # 用临时文件替换原文件
        shutil.move(temp_file_path, history_file)
        # 删除备份文件(可选)
        if os.path.exists(backup_file):
            os.remove(backup_file)
        print(f"删除了 {deleted_count} 条记录")
        return deleted_count
    except Exception as e:
        print(f"删除过程中出现错误: {e}")
        # 如果发生错误,恢复备份
        if 'backup_file' in locals() and os.path.exists(backup_file):
            print("尝试恢复备份...")
            try:
                shutil.move(backup_file, history_file)
                print("备份已恢复")
            except:
                print("恢复备份失败")
        # 清理临时文件
        if temp_file_path and os.path.exists(temp_file_path):
            try:
                os.remove(temp_file_path)
            except:
                pass
        return 0




#支持的查询操作符映射到对应的判断函数
QUERY_OPERATORS = {
    'eq': lambda item_value, query_value: item_value == query_value,
    'ne': lambda item_value, query_value: item_value != query_value,
    'lt': lambda item_value, query_value: item_value < query_value,
    'lte': lambda item_value, query_value: item_value <= query_value,
    'gt': lambda item_value, query_value: item_value > query_value,
    'gte': lambda item_value, query_value: item_value >= query_value,
    'in': lambda item_value, query_value: item_value in query_value,
    'like': lambda item_value, query_value: bool(re.search(str(query_value), str(item_value))),
}
MY_OP_KEYS = list(QUERY_OPERATORS.keys())
MY_OP_KEYS.append('or')


def check_condition(item: Dict[str, Any],query: Dict[str, Any]) -> bool:
    """检查单个记录是否满足所有查询条件"""
    # 处理其他AND条件(除了or)
    and_conditions = []
    # op 运算符 必须是 query_operators.key, 否则提示 KeyError
    for op in query.keys():
        if op not in MY_OP_KEYS:
            error = f"{op} 不是支持的查询操作符, 应该是,{json.dumps(MY_OP_KEYS, ensure_ascii=False)}"
            print(error)
            raise Exception(error)
    # 去掉 or  OR条件 后面 单独处理
    and_conditions = [op for op in query.keys() if op not in ['or']]
    if not and_conditions or  len(and_conditions)==0:
        # 避免 只有 or 的条件
        return False
    for op, conditions in query.items():
        if op == 'or':  # OR条件单独处理
            continue
        if not isinstance(conditions, dict):
            print(f"条件格式错误: {conditions}")
            continue
        for field, value in conditions.items():
            if field not in item:
                return False
            try:
                if not QUERY_OPERATORS[op](item[field], value):
                    return False
            except Exception as e:
                print(f"比较失败: {e}")
                # 如果比较失败(如类型不匹配),则视为不匹配
                return False
    #都没出错,说明匹配OK
    return True

def check_or_condition(item: Dict[str, Any],query: Dict[str, Any]) -> bool:
    """检查单个记录是否满足所有查询条件"""
    or_conditions =query.get('or', None)
    if or_conditions:
        if not isinstance(or_conditions, dict):
            print(f"OR条件格式错误: {or_conditions}")
            return False
        return check_condition(item,or_conditions)
    else:
        return False


def mutil_find(query: Dict[str, Any] = {}, limit: Optional[int] = None,
               history_file: str = HISTORY_FILE) -> List[Dict[str, Any]]:
    """
    查询所有匹配的历史记录 limit 限制,默认倒序
    Args:
        history_file: 历史记录文件路径
        query: 查询条件字典,支持以下操作符:
            - like: 正则匹配,如 {"name": "张.*"}
            - in: 在列表中,如 {"name": ["张三1", "张三2"]}
            - or: OR条件组合,支持多个条件
            - eq: 等于,如 {"name": "sea"}
            - ne: 不等于,如 {"name": "sea"}
            - lt: 小于,如 {"age": 18}
            - lte: 小于等于,如 {"age": 18}
            - gt: 大于,如 {"age": 18}
            - gte: 大于等于,如 {"age": 18}
            示例:
            {
                "like": {"name": "张"},  # 正则匹配
                "in": {"name": ["张三1", "张三2"]},
                "or": {"eq": {"ts": "2026-01-19 11:04:46"}},  # OR条件
                "eq": {"name": "sea"},
                "ne": {"name": "sea"},
                "lt": {"age": 50},
                "lte": {"age": 18},
                "gt": {"age": 50},
                "gte": {"age": 18}
            }
        limit: 限制查询多少条结果

    Returns:
        所有匹配的记录列表,按时间倒序排列(最新的在前)
    """
    results = []
    # 检查文件是否存在
    if not os.path.exists(history_file):
        print(f"文件不存在: {history_file}")
        return []
    try:
        #使用更简单的分块读取,从文件末尾开始
        with open(history_file, 'rb') as f:
            # 获取文件大小
            file_size = os.path.getsize(history_file)
            # 设置块大小(可根据实际情况调整)
            chunk_size = 1024 * 1024  # 1MB
            position = file_size
            buffer = b""
            lines_processed = 0
            max_lines_to_process = 10000  # 限制处理的行数,避免无限循环
            while position > 0 and lines_processed < max_lines_to_process:
                # 计算本次读取的起始位置
                read_size = min(chunk_size, position)
                position -= read_size
                # 移动到读取位置并读取数据
                f.seek(position)
                chunk = f.read(read_size)
                # 将新读取的数据添加到缓冲区前面
                buffer = chunk + buffer
                # 分割成行处理
                while b'\n' in buffer:
                    # 找到第一个完整行
                    newline_pos = buffer.find(b'\n')
                    if newline_pos == -1:
                        break
                    line_bytes = buffer[:newline_pos]
                    buffer = buffer[newline_pos + 1:]
                    if line_bytes:
                        try:
                            history = json.loads(line_bytes.decode('utf-8'))
                            lines_processed += 1
                            # 条件匹配
                            if check_condition(history,query):
                                results.append(history)
                                limit = limit-1
                                print(f"查询到匹配记录: {history}")
                                if limit <= 0:
                                    break
                            else:
                                # 避免数据重复
                                if check_or_condition(history,query):
                                    results.append(history)
                                    limit = limit-1
                                    print(f"or 查询到匹配记录: {history}")
                                    if limit <= 0:
                                        break
                        except json.JSONDecodeError:
                            # 可能是跨块的行,将数据放回缓冲区继续处理
                            buffer = line_bytes + b'\n' + buffer
                            break
                        except UnicodeDecodeError:
                            line_preview = line_bytes[:50] + b"..." if len(line_bytes) > 50 else line_bytes
                            print(f"编码错误,跳过行: {line_preview}")

                # 如果缓冲区过大,截断一部分
                if len(buffer) > chunk_size * 2:
                    buffer = buffer[-chunk_size:]
            # 处理缓冲区中剩余的数据(文件开头部分)
            if buffer:
                lines = buffer.split(b'\n')
                for line_bytes in lines:
                    if not line_bytes:
                        continue
                    try:
                        history = json.loads(line_bytes.decode('utf-8'))
                        lines_processed += 1
                        # 条件匹配
                        if check_condition(history,query):
                            results.append(history)
                            limit = limit-1
                            print(f"查询到匹配记录: {history}")
                            if limit <= 0:
                                break
                        else:
                            # 避免数据重复
                            if check_or_condition(history,query):
                                results.append(history)
                                limit = limit-1
                                print(f"or 查询到匹配记录: {history}")
                                if limit <= 0:
                                    break
                    except (json.JSONDecodeError, UnicodeDecodeError):
                        continue
    except Exception as e:
        traceback.print_exc()
        print(f"查询失败: {e}")
        return []
    print(f"共查询到 {len(results)} 条匹配记录")
    return results
    print(f"共查询到 {len(results)} 条匹配记录")
    return results





# 测试使用
def test_mutil_find():
    query1 = {
        "like": {"mark": "哈1"},
        "or": {"like": {"mark": "哈2"}},
        # "eq": {"name": "张三9"}
    }
    results = mutil_find(query=query1, limit=5)
    # 查询结果处理
    for result in results:
        print(f"找到记录: {result}")





def test_save_history():
    for i in range(10):
        save( {"name": "张三"+str(i), "age": i,"mark":"nihao \n 哈哈哈 \n 哈"})
        find_one( {"name": "张三"+str(i)})
        update({"name": "张三"+str(i)}, {"age": 198},5)


if __name__ == '__main__':
    # test_save_history()
    # delete({"age": 198}, 5)
    # find({'name': '张三5'},10)
    test_mutil_find()
    # test_save_history()

 

 

 

 

 

 

 

                       

  

posted on 2026-01-16 12:16  lshan  阅读(57)  评论(0)    收藏  举报