Fork me on GitHub

Spring AI

Ollama安装deepseek-r1:14b

显卡: RTX5080
Ollama下载: https://ollama.com/download
deepseek-r1:14b模型: https://ollama.com/library/deepseek-r1:14b
测试
image

配置Ollama允许网络访问

image

SpringAI配置

pom.xml

<?xml version="1.0" encoding="UTF-8"?>
<project xmlns="http://maven.apache.org/POM/4.0.0"
         xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
         xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 https://maven.apache.org/xsd/maven-4.0.0.xsd">
    <modelVersion>4.0.0</modelVersion>

    <parent>
        <groupId>org.springframework.boot</groupId>
        <artifactId>spring-boot-starter-parent</artifactId>
        <version>3.4.5</version>
        <relativePath/>
    </parent>

    <groupId>com.example</groupId>
    <artifactId>dual-chat-demo</artifactId>
    <version>0.0.1-SNAPSHOT</version>
    <name>dual-chat-demo</name>
    <description>SpringAI Demo:OpenAI 与 Ollama 双模型共存 + 基础对话</description>

    <properties>
        <java.version>21</java.version>
        <spring-ai.version>1.1.8</spring-ai.version>
    </properties>

    <dependencyManagement>
        <dependencies>
            <dependency>
                <groupId>org.springframework.ai</groupId>
                <artifactId>spring-ai-bom</artifactId>
                <version>${spring-ai.version}</version>
                <type>pom</type>
                <scope>import</scope>
            </dependency>
        </dependencies>
    </dependencyManagement>

    <dependencies>
        <dependency>
            <groupId>org.springframework.boot</groupId>
            <artifactId>spring-boot-starter-web</artifactId>
        </dependency>

        <!-- 两个模型 starter 同时引入:各自自动配置一个 ChatModel Bean,互不排除 -->
        <dependency>
            <groupId>org.springframework.ai</groupId>
            <artifactId>spring-ai-starter-model-openai</artifactId>
        </dependency>
        <dependency>
            <groupId>org.springframework.ai</groupId>
            <artifactId>spring-ai-starter-model-ollama</artifactId>
        </dependency>

        <!-- RAG 向量库支持:VectorStore + SimpleVectorStore -->
        <dependency>
            <groupId>org.springframework.ai</groupId>
            <artifactId>spring-ai-vector-store</artifactId>
        </dependency>
        <!-- RAG 检索 Advisor:QuestionAnswerAdvisor(独立于 vector-store,需单独引入) -->
        <dependency>
            <groupId>org.springframework.ai</groupId>
            <artifactId>spring-ai-advisors-vector-store</artifactId>
        </dependency>
        <!-- RAG 文档读取:Tika 解析 PDF / Word / Excel / PPT / TXT 等本地文件 -->
        <dependency>
            <groupId>org.springframework.ai</groupId>
            <artifactId>spring-ai-tika-document-reader</artifactId>
        </dependency>
    </dependencies>

    <build>
        <plugins>
            <plugin>
                <groupId>org.springframework.boot</groupId>
                <artifactId>spring-boot-maven-plugin</artifactId>
            </plugin>
        </plugins>
    </build>
</project>

application.yml

server:
  port: 8081
  servlet:
    encoding:
      charset: UTF-8
      force: true
      force-request: true
      force-response: true

spring:
  application:
    name: dual-chat-demo
  ai:
    openai:
      api-key: ${OPENAI_API_KEY:placeholder}   # 走 openai 路由时需要真实 key
      chat:
        options:
          model: gpt-4o-mini
          temperature: 0.7
    ollama:
      base-url: http://192.168.0.152:11434  #内网机器部署ollama
      chat:
        options:
          model: deepseek-r1:14b   #安装的模型
          temperature: 0.7
      # RAG 向量库用的 Embedding 模型,首次使用需先执行:ollama pull nomic-embed-text
      embedding:
        options:
          model: nomic-embed-text

SpringAI功能代码

ChatController基础对话和流式对话

import org.springframework.ai.chat.client.ChatClient;
import org.springframework.ai.chat.messages.UserMessage;
import org.springframework.ai.chat.model.ChatModel;
import org.springframework.ai.chat.model.ChatResponse;
import org.springframework.ai.chat.prompt.Prompt;
import org.springframework.ai.ollama.OllamaChatModel;
import org.springframework.ai.openai.OpenAiChatModel;
import org.springframework.http.MediaType;
import org.springframework.util.StringUtils;
import org.springframework.web.bind.annotation.GetMapping;
import org.springframework.web.bind.annotation.RequestMapping;
import org.springframework.web.bind.annotation.RequestParam;
import org.springframework.web.bind.annotation.RestController;
import reactor.core.publisher.Flux;

/**
 * 方案A核心:不注入 ChatModel 接口,而是注入两个【具体类型】,
 * 容器里同时存在 OpenAiChatModel 和 OllamaChatModel 两个 Bean,
 * 按类型注入天然无歧义 —— 这就是「多模型共存」的全部秘密。
 */
@RestController
@RequestMapping("/api/chat")
public class ChatController {

    // ==================== 依赖注入 ====================

    private final OpenAiChatModel openAiChatModel;
    private final OllamaChatModel ollamaChatModel;

    public ChatController(OpenAiChatModel openAiChatModel,
                          OllamaChatModel ollamaChatModel) {
        this.openAiChatModel = openAiChatModel;
        this.ollamaChatModel = ollamaChatModel;
    }

    // ==================== 阻塞式对话 ====================

    /**
     * 基础对话(阻塞式):?model=openai|ollama&message=xxx
     *   GET http://localhost:8081/api/chat/ask?model=ollama&message=用一句话介绍Spring AI
     */
    @GetMapping("/ask")
    public String ask(@RequestParam(defaultValue = "ollama") String model,
                      @RequestParam(defaultValue = "用一句话介绍 Spring AI") String message) {
        ChatResponse response = resolve(model).call(new Prompt(new UserMessage(message)));
        return response.getResult().getOutput().getText();
    }

    // ==================== 流式对话(SSE) ====================

    /**
     * 流式对话(SSE,底层 ChatModel.stream 调用):逐 token 返回,末尾追加 [DONE]。
     *
     * <p>【教学 / 兜底写法】日常请优先用下面的 ChatClient 版本。
     *
     * <p>与 ChatClient 版的本质区别:这里拿到的是 {@code Flux<ChatResponse>},能接触到
     * 响应对象本身,而不只是文本。因此以下场景必须用它,无法被 ChatClient 替代:
     * <ol>
     *   <li>统计 / 计费 token 用量 —— {@code ChatResponse.getMetadata().getUsage()}</li>
     *   <li>依据 finishReason 做分支 —— LENGTH 截断提示、TOOL_CALL 走工具调用、SAFETY 拦截</li>
     *   <li>Function Calling / Tool 调用 —— ToolCallingChatOptions、tool result 回填</li>
     *   <li>手动编排 Prompt —— 多轮 System/User/Assistant/ToolResponseMessage、多模态 Media 消息</li>
     *   <li>处理 chunk 级中间态 —— 如模型的 reasoningContent(思考过程)</li>
     * </ol>
     * 只需要纯文本、或想挂 Advisor(记忆 / RAG / 日志)时,请用 {@link #streamByChatClient}。
     *
     * <p>注意:URL 中的中文需百分号编码,否则 Tomcat 会直接返回 400,请求到不了这里。
     *   curl -N "http://localhost:8081/api/chat/stream?message=%E8%AE%B2%E4%B8%AA%E7%AC%91%E8%AF%9D"
     * <br>或交由 curl 编码(UTF-8 终端下):
     *   curl -N -G "http://localhost:8081/api/chat/stream" --data-urlencode "message=讲个笑话"
     */
    @GetMapping(value = "/stream", produces = MediaType.TEXT_EVENT_STREAM_VALUE)
    public Flux<String> stream(@RequestParam(defaultValue = "ollama") String model,
                               @RequestParam(defaultValue = "用一句话介绍 Spring AI") String message) {
        return resolve(model).stream(new Prompt(new UserMessage(message)))
                .map(resp -> resp.getResult().getOutput().getText())
                .filter(StringUtils::hasText)       // 过滤掉心跳/空片段
                .concatWith(Flux.just("[DONE]"));   // 结束标记,便于前端判断
    }

    /**
     * 流式对话(SSE,高层 ChatClient.stream):写法更简洁,后续可接 Advisor(记忆/RAG)。
     *   curl -N "http://localhost:8081/api/chat/stream/client?model=openai&message=讲个笑话"
     */
    @GetMapping(value = "/stream/client", produces = MediaType.TEXT_EVENT_STREAM_VALUE)
    public Flux<String> streamByChatClient(@RequestParam(defaultValue = "ollama") String model,
                                           @RequestParam(defaultValue = "用一句话介绍 Spring AI") String message) {
        return ChatClient.create(resolve(model))
                .prompt().user(message)
                .stream().content()
                .filter(StringUtils::hasText);
    }

    // ==================== 私有方法 ====================

    private ChatModel resolve(String model) {
        return "openai".equalsIgnoreCase(model) ? openAiChatModel : ollamaChatModel;
    }
}

测试基础对话: http://localhost:8081/api/chat/ask?model=openai&message=用一句话介绍Spring
image

测试流式对话: http://localhost:8081/api/chat/stream?model=ollama&message=讲个笑话
image

模板输入 Prompt 与结构化输出

// ==================== 模板输入 Prompt 与结构化输出 ====================

    /**
     * 模板 Prompt:用 {composer} 占位,运行时 param 注入,提示词可配置/可版本化(大纲阶段二)。
     *   curl -G "http://localhost:8081/api/chat/template?model=ollama&composer=%E4%B9%85%E7%9F%B3%E8%AE%A9"
     * <br>或交由 curl 编码(UTF-8 终端下):
     *   curl -N -G "http://localhost:8081/api/chat/template" --data-urlencode "composer=久石让"
     */
    @GetMapping("/template")
    public String template(@RequestParam(defaultValue = "ollama") String model,
                           @RequestParam(defaultValue = "久石让") String composer) {
        return ChatClient.create(resolve(model))
                .prompt()
                .user(u -> u.text("请列出 5 部由 {composer} 配乐的经典电影,直接给出列表。")
                        .param("composer", composer))
                .call()
                .content();
    }

    /**
     * 结构化输出:模型返回直接映射为 Java 类型(record),自动完成 JSON Schema 生成与解析(大纲阶段二)。
     * 字段说明写在 {@link ActorFilms} 的 {@code @JsonPropertyDescription} 里,改那里即可调整输出格式。
     *   curl -G "http://localhost:8081/api/chat/movies?model=openai&actor=%E6%A2%81%E6%9C%9D%E4%BC%9F"
     * <br>或交由 curl 编码(UTF-8 终端下):
     *   curl -N -G "http://localhost:8081/api/chat/movies" --data-urlencode "actor=梁朝伟"
     *
     * <p>注意:推理型模型(如 deepseek-r1)的思考内容可能干扰 JSON 解析,报错时换非推理模型
     * (如 qwen2.5)或走 openai 路由。
     */
    @GetMapping("/movies")
    public ActorFilms movies(@RequestParam(defaultValue = "ollama") String model,
                             @RequestParam(defaultValue = "梁朝伟") String actor) {
        return ChatClient.create(resolve(model))
                .prompt()
                .user(u -> u.text("生成演员 {actor} 的 3~5 部代表作," 
                        + "每部给出片名、上映年份、类型、导演、评分和一句话推荐理由。")
                        .param("actor", actor))
                .call()
                .entity(ActorFilms.class);
    }

结构化输出

import com.fasterxml.jackson.annotation.JsonPropertyDescription;

import java.util.List;

/**
 * 结构化输出用的目标类型(大纲阶段二)。
 *
 * <p>record 自动提供访问器,Spring AI 据此生成 JSON Schema 并塞进 Prompt,
 * 模型按 Schema 输出 JSON 后自动反序列化。
 *
 * <p>关键点:{@code @JsonPropertyDescription} 会写进生成的 JSON Schema 的 description 字段,
 * 这是告诉模型「每个字段要什么格式」的唯一手段 —— 描述越具体(尤其是单位、取值范围),
 * 弱模型的表现越稳。改动这里的描述就等于改提示词,无需动 Controller。
 *
 * <p>嵌套 record 会生成嵌套对象 Schema,因此可以直接表达「一个演员对应多部电影」这种一对多结构。
 */
public record ActorFilms(
        @JsonPropertyDescription("演员姓名,与输入保持一致") String actor,
        @JsonPropertyDescription("代表作列表,3~5 部,按上映年份升序排列") List<Movie> movies) {

    public record Movie(
            @JsonPropertyDescription("电影片名") String title,
            @JsonPropertyDescription("上映年份,四位数字,如 1994") Integer year,
            @JsonPropertyDescription("电影类型,如 剧情 / 爱情 / 战争 / 科幻") String genre,
            @JsonPropertyDescription("导演姓名") String director,
            @JsonPropertyDescription("豆瓣评分,0~10 的浮点数,保留一位小数") Double rating,
            @JsonPropertyDescription("一句话推荐理由,25 字以内") String highlight) {
    }
}

测试模板输入: http://localhost:8081/api/chat/template?model=ollama&composer=久石让
image

测试结构化输出: http://localhost:8081/api/chat/movies?model=ollama&actor=梁朝伟
image

多轮记忆对话(SSE)

    // ==================== 多轮记忆对话(SSE) ====================

    /**
     * 多轮记忆流式对话(SSE):同一 conversationId 会自动带上历史上下文。
     * 图片示例:message="我叫小明,记得我",conversationId="user-001"。
     *
     * <p>POST 版本:前端用 fetch + ReadableStream 请求,支持自定义 Header。
     *   curl -N -X POST "http://localhost:8081/api/chat/memory" \\
     *     -d "model=ollama" \\
     *     -d "conversationId=user-001" \\
     *     --data-urlencode "message=我叫小明,记得我"
     */
    @PostMapping(value = "/memory", produces = MediaType.TEXT_EVENT_STREAM_VALUE + ";charset=UTF-8")
    public Flux<String> memoryByPost(@RequestParam(defaultValue = "ollama") String model,
                                     @RequestParam String message,
                                     @RequestParam String conversationId) {
        return doMemoryChat(model, message, conversationId);
    }

    /**
     * 多轮记忆流式对话(SSE):GET 版本,方便直接用 EventSource 测试。
     *   curl -N -G "http://localhost:8081/api/chat/memory" --data-urlencode "message=我叫小明,记得我" -d "conversationId=user-001"
     */
    @GetMapping(value = "/memory", produces = MediaType.TEXT_EVENT_STREAM_VALUE + ";charset=UTF-8")
    public Flux<String> memoryByGet(@RequestParam(defaultValue = "ollama") String model,
                                    @RequestParam String message,
                                    @RequestParam String conversationId) {
        return doMemoryChat(model, message, conversationId);
    }

    private Flux<String> doMemoryChat(String model, String message, String conversationId) {
        return ChatClient.create(resolve(model))
                .prompt()
                .user(message)
                // 【记忆核心】Advisor = 引擎(负责读写历史),param = 钥匙(指定会话 ID),两者缺一不可
                .advisors(a -> a
                        // 挂载记忆拦截器:请求前把该会话历史拼进 prompt,响应后把本轮问答写回 ChatMemory
                        .advisors(MessageChatMemoryAdvisor.builder(chatMemory).build())
                        // 告知 Advisor 去哪个会话读写;不传会回退默认 ID,导致所有会话互相串台
                        .param(ChatMemory.CONVERSATION_ID, conversationId))
                .stream()
                .content()
                .filter(StringUtils::hasText)
                .concatWith(Flux.just("[DONE]"));
    }

测试(前几次89轮隔着都能记录,后面出现上面刚告诉下一轮问就不知道的情况)
image

image
原因分析:
deepseek-r1:14b,这是一个推理模型。这类模型输出本身就有随机性,即使上下文完全一样,两次回答也可能不同。尤其是 14b 这种小尺寸蒸馏版.

工具调用

    // ==================== Function Calling / 工具调用 ====================

    /**
     * 工具调用:模型自动判断是否需要调用 LightingTools 控制灯光。
     * 示例:message="把客厅的灯调暗到 30%"
     *
     * <p>POST http://localhost:8081/api/chat/tool
     * curl -X POST "http://localhost:8081/api/chat/tool" -d "model=openai" --data-urlencode "message=把客厅的灯调暗到 30%"
     */
    @PostMapping("/tool")
    public String tool(@RequestParam(defaultValue = "openai") String model,
                       @RequestParam String message) {
        return ChatClient.create(resolve(model))
                .prompt()
                .user(message)
                .tools(lightingTools)
                .call()
                .content();
    }

工具类

import java.util.Map;
import java.util.concurrent.ConcurrentHashMap;
import org.springframework.ai.tool.annotation.Tool;
import org.springframework.ai.tool.annotation.ToolParam;
import org.springframework.stereotype.Component;

/**
 * 智能家居灯光控制工具(大纲阶段五:Function Calling)。
 *
 * <p>被 {@code @Tool} 标注的方法会被 Spring AI 自动注册为可调用的函数,
 * 模型根据用户意图决定是否调用、并自动填充参数 —— 一个类里可以写多个工具方法,
 * 全部会被扫描到,Controller 端无需逐个声明。
 *
 * <p>这里用一个内存 Map 模拟设备状态,方便看到「开灯 → 调暗 → 关灯」的真实效果;
 * 生产环境换成智能家居 SDK / MQTT / Home Assistant 调用即可。
 */
@Component
public class LightingTools {

    /** 房间 → 当前状态(on / brightness),内存模拟,重启丢失 */
    private final Map<String, LightState> states = new ConcurrentHashMap<>();

    /**
     * 工具 1:开关灯。
     *
     * @param room 房间名称,如 客厅、卧室
     * @param on   true=开灯,false=关灯
     * @return 操作结果描述
     */
    @Tool(name = "switch_light", description = "打开或关闭指定房间的灯")
    public String switchLight(
            @ToolParam(required = true, description = "房间名称,例如 客厅、卧室") String room,
            @ToolParam(required = true, description = "true 表示开灯,false 表示关灯") boolean on) {

        LightState state = states.computeIfAbsent(room, r -> new LightState());
        state.on = on;
        System.out.println("[LightingTools] " + room + " 的灯已" + (on ? "打开" : "关闭"));
        return "已把" + room + "的灯" + (on ? "打开" : "关闭");
    }

    /**
     * 工具 2:调亮度。
     *
     * @param room       房间名称,如 客厅、卧室
     * @param brightness 亮度百分比,0~100 的整数
     * @return 操作结果描述
     */
    @Tool(name = "adjust_light_brightness", description = "调整指定房间的灯光亮度")
    public String adjustBrightness(
            @ToolParam(required = true, description = "房间名称,例如 客厅、卧室") String room,
            @ToolParam(required = true, description = "亮度百分比,0~100 的整数") int brightness) {

        // 参数校验
        if (brightness < 0 || brightness > 100) {
            return "亮度必须在 0~100 之间,当前传入:" + brightness;
        }

        LightState state = states.computeIfAbsent(room, r -> new LightState());
        // 调亮度隐含开灯
        state.on = true;
        state.brightness = brightness;
        System.out.println("[LightingTools] 已将 " + room + " 的灯光亮度调整为 " + brightness + "%");
        return "已把" + room + "的灯调到 " + brightness + "%";
    }

    /**
     * 工具 3:查询当前灯光状态(可选,方便模型回答「客厅灯现在亮着吗」这类问题)。
     *
     * @param room 房间名称,如 客厅、卧室
     * @return 该房间灯光的当前状态
     */
    @Tool(name = "query_light_status", description = "查询指定房间灯光的开关状态和亮度")
    public String queryStatus(
            @ToolParam(required = true, description = "房间名称,例如 客厅、卧室") String room) {

        LightState state = states.get(room);
        if (state == null) {
            return room + " 暂无灯光设备";
        }
        return state.on ? room + "的灯是开着的,亮度 " + state.brightness + "%" : room + "的灯是关着的";
    }

    /** 单个房间的灯光状态 */
    private static class LightState {
        private boolean on = false;
        private int brightness = 100;
    }
}

测试: curl.exe -X POST "http://localhost:8081/api/chat/tool" -d "model=ollama" --data-urlencode "message=把客厅的灯打开"
image
deepseek-r1:14b 是推理模型不能使用 工具调用

模型 工具支持
qwen2.5 / qwen2.5:14b 很好
llama3.1 / llama3.1:8b 官方支持
mistral / mistral:7b 支持
phi4 支持
deepseek-r1:14b 不支持 / 极不稳定

换成qwen2.5调用成功
image

RAG检索配置

pom.xml添加依赖

        <!-- RAG 向量库支持:VectorStore + SimpleVectorStore -->
        <dependency>
            <groupId>org.springframework.ai</groupId>
            <artifactId>spring-ai-vector-store</artifactId>
        </dependency>
        <!-- RAG 检索 Advisor:QuestionAnswerAdvisor(独立于 vector-store,需单独引入) -->
        <dependency>
            <groupId>org.springframework.ai</groupId>
            <artifactId>spring-ai-advisors-vector-store</artifactId>
        </dependency>
        <!-- RAG 文档读取:Tika 解析 PDF / Word / Excel / PPT / TXT 等本地文件 -->
        <dependency>
            <groupId>org.springframework.ai</groupId>
            <artifactId>spring-ai-tika-document-reader</artifactId>
        </dependency>

配置文件

  application:
    name: SpringAI-chat
  ai:
    ollama:
      base-url: http://192.168.0.152:11434
      chat:
        options:
          model: qwen2.5:14b #deepseek-r1:14b
          temperature: 0.7
      # RAG 向量库用的 Embedding 模型,首次使用需先执行:ollama pull nomic-embed-text
      embedding:
        options:
          model: nomic-embed-text

# 自定义配置放最后:RAG 知识库文档目录(支持 PDF / Word / Excel / PPT / TXT)
# 把公司文档丢进该目录,启动时自动解析并向量化;目录为空则回退内置示例
rag:
  documents-path: ./docs

RagConfig向量配置

import java.io.IOException;
import java.nio.file.Files;
import java.nio.file.Path;
import java.nio.file.Paths;
import java.util.ArrayList;
import java.util.List;
import java.util.stream.Stream;
import org.springframework.ai.document.Document;
import org.springframework.ai.ollama.OllamaEmbeddingModel;
import org.springframework.ai.reader.tika.TikaDocumentReader;
import org.springframework.ai.transformer.splitter.TokenTextSplitter;
import org.springframework.ai.vectorstore.SimpleVectorStore;
import org.springframework.ai.vectorstore.VectorStore;
import org.springframework.beans.factory.annotation.Value;
import org.springframework.context.annotation.Bean;
import org.springframework.context.annotation.Configuration;
import org.springframework.core.io.FileSystemResource;

/**
 * RAG 检索增强配置(大纲阶段四):读取本地公司文档 → 切块 → 写入向量库。
 *
 * <p>文档目录由 {@code rag.documents-path} 指定,默认 {@code ./docs}。
 * 把 PDF / Word / Excel / PPT / TXT 等文件直接丢进该目录即可,应用启动时自动加载。
 *
 * <p>解析用 Apache Tika({@link TikaDocumentReader}),能自动识别 1000+ 种格式,
 * 无需为每种后缀写不同代码。若只需要读 PDF,可换成轻量得多的 {@code PagePdfDocumentReader}。
 *
 * <p>长文档必须先用 {@link TokenTextSplitter} 切块:整篇塞进 prompt 会超出模型上下文窗口,
 * 且检索粒度太粗(一大段里只有一句相关也会整段召回),切块后检索精度才高。
 *
 * <p>目录为空时回退到内置示例文档,保证 /api/chat/rag/ask 始终可演示。
 */
@Configuration
public class RagConfig {

    /**
     * 本地公司文档目录。支持绝对路径(如 D:/company-docs)或相对项目根目录的路径。
     */
    @Value("${rag.documents-path:./docs}")
    private String documentsPath;

    @Bean
    public VectorStore vectorStore(OllamaEmbeddingModel ollamaEmbeddingModel) {
        SimpleVectorStore vectorStore = SimpleVectorStore.builder(ollamaEmbeddingModel).build();

        List<Document> chunks = loadFromDocuments();
        if (chunks.isEmpty()) {
            System.out.println("[RagConfig] 未读到本地文档,回退到内置示例知识库");
            chunks = defaultDocuments();
        }
        vectorStore.add(chunks);
        System.out.println("[RagConfig] 知识库加载完成,共 " + chunks.size() + " 个文本块");
        return vectorStore;
    }

    /**
     * 遍历文档目录,逐个用 Tika 解析,再统一切块。
     */
    private List<Document> loadFromDocuments() {
        Path dir = Paths.get(documentsPath);
        if (!Files.isDirectory(dir)) {
            System.out.println("[RagConfig] 文档目录不存在,跳过: " + dir.toAbsolutePath());
            return List.of();
        }

        List<Document> raw = new ArrayList<>();
        try (Stream<Path> files = Files.walk(dir)) {
            List<Path> fileList = files
                    .filter(Files::isRegularFile)
                    // 跳过隐藏文件(如 macOS 的 .DS_Store)
                    .filter(p -> !p.getFileName().toString().startsWith("."))
                    .toList();

            for (Path file : fileList) {
                try {
                    TikaDocumentReader reader = new TikaDocumentReader(new FileSystemResource(file));
                    List<Document> docs = reader.get();
                    raw.addAll(docs);
                    System.out.println("[RagConfig] 已读取: " + file.getFileName() + " -> " + docs.size() + " 个文档");
                } catch (Exception e) {
                    // 单个文件解析失败不影响其他文件
                    System.out.println("[RagConfig] 读取失败已跳过: " + file.getFileName() + " -> " + e.getMessage());
                }
            }
        } catch (IOException e) {
            System.out.println("[RagConfig] 遍历目录失败: " + e.getMessage());
            return List.of();
        }

        if (raw.isEmpty()) {
            return List.of();
        }
        // 切块:默认约 800 token/块,相邻块有重叠以保留上下文连贯性
        return new TokenTextSplitter().apply(raw);
    }

    /**
     * 内置示例知识库:docs 目录为空时使用,仅用于演示,接了真实文档后自然会走 {@link #loadFromDocuments()}。
     */
    private List<Document> defaultDocuments() {
        return List.of(
                new Document("公司年假政策:员工入职满一年后,每年可享受 10 天带薪年假。"),
                new Document("公司年假政策:入职不满一年的员工,按实际工作月份折算年假天数,每月 0.83 天。")
        );
    }
}

image

测试
image

Agent Chain 链式编排


    // ==================== Agent Chain 链式编排 ====================

    /**
     * 两步链式编排:先出要点,再据要点扩写成文(大纲阶段六)。
     *
     * <p>这是最基础、也最实用的「工作流 / Agent 链」形态 —— 把一个复杂任务拆成两次
     * 顺序调用,<b>上一步的输出作为下一步的输入</b>,而不是让模型一步到位:
     * <ol>
     *   <li><b>第一步(出要点)</b>:让模型先做「策划」,只输出主题下的 N 条要点大纲,
     *       任务单一、约束明确,弱模型也能稳定完成;</li>
     *   <li><b>第二步(扩写)</b>:把第一步得到的要点原样喂回,让模型「照着提纲写」,
     *       把碎片要点扩写成结构完整、语言流畅的文章。</li>
     * </ol>
     *
     * <p>为什么拆成两步,而不是一个 Prompt 全包?—— 拆解后每一步的 Prompt 更聚焦,
     * 模型所需的「注意力」更小,输出质量与稳定性都更好;也便于在中间步骤插入
     * 校验、改写、检索(RAG)等增强逻辑,这正是 Agent / 工作流编排的核心思路。
     *
     * <p>图片示例:topic="量子计算"。
     *
     * <p>一个入口同时支持 GET 与 POST(见下面 {@code method} 属性):
     * <ul>
     *   <li><b>GET</b> —— 参数都有默认值,浏览器地址栏直接打开就能看效果:
     *       <pre>http://localhost:8081/api/agent/chain</pre>
     *       curl -G "http://localhost:8081/api/agent/chain" --data-urlencode "topic=量子计算"</li>
     *   <li><b>POST</b> —— 前端表单 / fetch 调用:
     *       curl -X POST "http://localhost:8081/api/agent/chain" -d "model=ollama" --data-urlencode "topic=量子计算"</li>
     * </ul>
     *
     * <p>注意:URL 中的中文需百分号编码,否则部分浏览器/代理可能返回 400;
     * 直接把中文粘进地址栏时浏览器会自动编码,一般无需手动处理。
     *
     * <p>返回 {@link ChainResult}(JSON):同时带出中间产物 {@code outline}(要点)与最终
     * {@code article}(扩写),便于前端分段展示,也能直观看到「链」上每一步做了什么。
     * 浏览器里会挤成一整行,装了 JSON 格式化插件或换成 {@code curl | jq} 更好读。
     *
     * <p>为什么写成 {@code @RequestMapping(method = {GET, POST})} 而不是
     * {@code @GetMapping + @PostMapping} 两个方法:语义上是同一个「链」入口、逻辑完全一致,
     * 合并后只有一份签名与一份 Javadoc,避免改一处漏一处。
     */
    @RequestMapping(value = "/chain", method = {RequestMethod.GET, RequestMethod.POST})
    public ChainResult chain(@RequestParam(defaultValue = "ollama") String model,
                             @RequestParam(defaultValue = "量子计算") String topic) {
        return doChain(model, topic);
    }

    private ChainResult doChain(String model, String topic) {
        ChatClient client = ChatClient.create(resolve(model));

        // ---- 第一步:出要点(把「策划」这一步单独交给模型,任务聚焦、输出稳定)----
        String outline = client.prompt()
                .user(u -> u.text("你是内容策划。请针对主题「{topic}」列出 5 条要点大纲,"
                        + "每条一句话,直接输出编号列表,不要额外的开场白或解释。")
                        .param("topic", topic))
                .call()
                .content();

        // ---- 第二步:把上一步的要点作为输入,扩写成文(链式编排的关键:串起两次调用)----
        String article = client.prompt()
                .user(u -> u.text("请根据以下要点,扩写成一篇 300 字左右、结构完整、语言流畅的短文。"
                        + "主题是「{topic}」,要点只是骨架,请自行补充衔接与过渡,不要只是罗列要点。\n\n"
                        + "【要点】\n{outline}")
                        .param("topic", topic)
                        .param("outline", outline))
                .call()
                .content();

        return new ChainResult(topic, outline, article);
    }

    /**
     * Agent 链式编排的返回体:暴露中间产物与最终结果,方便前端分步展示。
     *
     * @param topic   原始输入主题
     * @param outline 第一步产出的要点大纲(中间产物)
     * @param article 第二步据要点扩写出的正文(最终结果)
     */
    public record ChainResult(String topic, String outline, String article) {
    }

测试
image

posted @ 2026-09-28 15:59  秋夜雨巷  阅读(6)  评论(0)    收藏  举报