Spring AI
Ollama安装deepseek-r1:14b
显卡: RTX5080
Ollama下载: https://ollama.com/download
deepseek-r1:14b模型: https://ollama.com/library/deepseek-r1:14b
测试

配置Ollama允许网络访问

SpringAI配置
pom.xml
<?xml version="1.0" encoding="UTF-8"?>
<project xmlns="http://maven.apache.org/POM/4.0.0"
xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 https://maven.apache.org/xsd/maven-4.0.0.xsd">
<modelVersion>4.0.0</modelVersion>
<parent>
<groupId>org.springframework.boot</groupId>
<artifactId>spring-boot-starter-parent</artifactId>
<version>3.4.5</version>
<relativePath/>
</parent>
<groupId>com.example</groupId>
<artifactId>dual-chat-demo</artifactId>
<version>0.0.1-SNAPSHOT</version>
<name>dual-chat-demo</name>
<description>SpringAI Demo:OpenAI 与 Ollama 双模型共存 + 基础对话</description>
<properties>
<java.version>21</java.version>
<spring-ai.version>1.1.8</spring-ai.version>
</properties>
<dependencyManagement>
<dependencies>
<dependency>
<groupId>org.springframework.ai</groupId>
<artifactId>spring-ai-bom</artifactId>
<version>${spring-ai.version}</version>
<type>pom</type>
<scope>import</scope>
</dependency>
</dependencies>
</dependencyManagement>
<dependencies>
<dependency>
<groupId>org.springframework.boot</groupId>
<artifactId>spring-boot-starter-web</artifactId>
</dependency>
<!-- 两个模型 starter 同时引入:各自自动配置一个 ChatModel Bean,互不排除 -->
<dependency>
<groupId>org.springframework.ai</groupId>
<artifactId>spring-ai-starter-model-openai</artifactId>
</dependency>
<dependency>
<groupId>org.springframework.ai</groupId>
<artifactId>spring-ai-starter-model-ollama</artifactId>
</dependency>
<!-- RAG 向量库支持:VectorStore + SimpleVectorStore -->
<dependency>
<groupId>org.springframework.ai</groupId>
<artifactId>spring-ai-vector-store</artifactId>
</dependency>
<!-- RAG 检索 Advisor:QuestionAnswerAdvisor(独立于 vector-store,需单独引入) -->
<dependency>
<groupId>org.springframework.ai</groupId>
<artifactId>spring-ai-advisors-vector-store</artifactId>
</dependency>
<!-- RAG 文档读取:Tika 解析 PDF / Word / Excel / PPT / TXT 等本地文件 -->
<dependency>
<groupId>org.springframework.ai</groupId>
<artifactId>spring-ai-tika-document-reader</artifactId>
</dependency>
</dependencies>
<build>
<plugins>
<plugin>
<groupId>org.springframework.boot</groupId>
<artifactId>spring-boot-maven-plugin</artifactId>
</plugin>
</plugins>
</build>
</project>
application.yml
server:
port: 8081
servlet:
encoding:
charset: UTF-8
force: true
force-request: true
force-response: true
spring:
application:
name: dual-chat-demo
ai:
openai:
api-key: ${OPENAI_API_KEY:placeholder} # 走 openai 路由时需要真实 key
chat:
options:
model: gpt-4o-mini
temperature: 0.7
ollama:
base-url: http://192.168.0.152:11434 #内网机器部署ollama
chat:
options:
model: deepseek-r1:14b #安装的模型
temperature: 0.7
# RAG 向量库用的 Embedding 模型,首次使用需先执行:ollama pull nomic-embed-text
embedding:
options:
model: nomic-embed-text
SpringAI功能代码
ChatController基础对话和流式对话
import org.springframework.ai.chat.client.ChatClient;
import org.springframework.ai.chat.messages.UserMessage;
import org.springframework.ai.chat.model.ChatModel;
import org.springframework.ai.chat.model.ChatResponse;
import org.springframework.ai.chat.prompt.Prompt;
import org.springframework.ai.ollama.OllamaChatModel;
import org.springframework.ai.openai.OpenAiChatModel;
import org.springframework.http.MediaType;
import org.springframework.util.StringUtils;
import org.springframework.web.bind.annotation.GetMapping;
import org.springframework.web.bind.annotation.RequestMapping;
import org.springframework.web.bind.annotation.RequestParam;
import org.springframework.web.bind.annotation.RestController;
import reactor.core.publisher.Flux;
/**
* 方案A核心:不注入 ChatModel 接口,而是注入两个【具体类型】,
* 容器里同时存在 OpenAiChatModel 和 OllamaChatModel 两个 Bean,
* 按类型注入天然无歧义 —— 这就是「多模型共存」的全部秘密。
*/
@RestController
@RequestMapping("/api/chat")
public class ChatController {
// ==================== 依赖注入 ====================
private final OpenAiChatModel openAiChatModel;
private final OllamaChatModel ollamaChatModel;
public ChatController(OpenAiChatModel openAiChatModel,
OllamaChatModel ollamaChatModel) {
this.openAiChatModel = openAiChatModel;
this.ollamaChatModel = ollamaChatModel;
}
// ==================== 阻塞式对话 ====================
/**
* 基础对话(阻塞式):?model=openai|ollama&message=xxx
* GET http://localhost:8081/api/chat/ask?model=ollama&message=用一句话介绍Spring AI
*/
@GetMapping("/ask")
public String ask(@RequestParam(defaultValue = "ollama") String model,
@RequestParam(defaultValue = "用一句话介绍 Spring AI") String message) {
ChatResponse response = resolve(model).call(new Prompt(new UserMessage(message)));
return response.getResult().getOutput().getText();
}
// ==================== 流式对话(SSE) ====================
/**
* 流式对话(SSE,底层 ChatModel.stream 调用):逐 token 返回,末尾追加 [DONE]。
*
* <p>【教学 / 兜底写法】日常请优先用下面的 ChatClient 版本。
*
* <p>与 ChatClient 版的本质区别:这里拿到的是 {@code Flux<ChatResponse>},能接触到
* 响应对象本身,而不只是文本。因此以下场景必须用它,无法被 ChatClient 替代:
* <ol>
* <li>统计 / 计费 token 用量 —— {@code ChatResponse.getMetadata().getUsage()}</li>
* <li>依据 finishReason 做分支 —— LENGTH 截断提示、TOOL_CALL 走工具调用、SAFETY 拦截</li>
* <li>Function Calling / Tool 调用 —— ToolCallingChatOptions、tool result 回填</li>
* <li>手动编排 Prompt —— 多轮 System/User/Assistant/ToolResponseMessage、多模态 Media 消息</li>
* <li>处理 chunk 级中间态 —— 如模型的 reasoningContent(思考过程)</li>
* </ol>
* 只需要纯文本、或想挂 Advisor(记忆 / RAG / 日志)时,请用 {@link #streamByChatClient}。
*
* <p>注意:URL 中的中文需百分号编码,否则 Tomcat 会直接返回 400,请求到不了这里。
* curl -N "http://localhost:8081/api/chat/stream?message=%E8%AE%B2%E4%B8%AA%E7%AC%91%E8%AF%9D"
* <br>或交由 curl 编码(UTF-8 终端下):
* curl -N -G "http://localhost:8081/api/chat/stream" --data-urlencode "message=讲个笑话"
*/
@GetMapping(value = "/stream", produces = MediaType.TEXT_EVENT_STREAM_VALUE)
public Flux<String> stream(@RequestParam(defaultValue = "ollama") String model,
@RequestParam(defaultValue = "用一句话介绍 Spring AI") String message) {
return resolve(model).stream(new Prompt(new UserMessage(message)))
.map(resp -> resp.getResult().getOutput().getText())
.filter(StringUtils::hasText) // 过滤掉心跳/空片段
.concatWith(Flux.just("[DONE]")); // 结束标记,便于前端判断
}
/**
* 流式对话(SSE,高层 ChatClient.stream):写法更简洁,后续可接 Advisor(记忆/RAG)。
* curl -N "http://localhost:8081/api/chat/stream/client?model=openai&message=讲个笑话"
*/
@GetMapping(value = "/stream/client", produces = MediaType.TEXT_EVENT_STREAM_VALUE)
public Flux<String> streamByChatClient(@RequestParam(defaultValue = "ollama") String model,
@RequestParam(defaultValue = "用一句话介绍 Spring AI") String message) {
return ChatClient.create(resolve(model))
.prompt().user(message)
.stream().content()
.filter(StringUtils::hasText);
}
// ==================== 私有方法 ====================
private ChatModel resolve(String model) {
return "openai".equalsIgnoreCase(model) ? openAiChatModel : ollamaChatModel;
}
}
测试基础对话: http://localhost:8081/api/chat/ask?model=openai&message=用一句话介绍Spring

测试流式对话: http://localhost:8081/api/chat/stream?model=ollama&message=讲个笑话

模板输入 Prompt 与结构化输出
// ==================== 模板输入 Prompt 与结构化输出 ====================
/**
* 模板 Prompt:用 {composer} 占位,运行时 param 注入,提示词可配置/可版本化(大纲阶段二)。
* curl -G "http://localhost:8081/api/chat/template?model=ollama&composer=%E4%B9%85%E7%9F%B3%E8%AE%A9"
* <br>或交由 curl 编码(UTF-8 终端下):
* curl -N -G "http://localhost:8081/api/chat/template" --data-urlencode "composer=久石让"
*/
@GetMapping("/template")
public String template(@RequestParam(defaultValue = "ollama") String model,
@RequestParam(defaultValue = "久石让") String composer) {
return ChatClient.create(resolve(model))
.prompt()
.user(u -> u.text("请列出 5 部由 {composer} 配乐的经典电影,直接给出列表。")
.param("composer", composer))
.call()
.content();
}
/**
* 结构化输出:模型返回直接映射为 Java 类型(record),自动完成 JSON Schema 生成与解析(大纲阶段二)。
* 字段说明写在 {@link ActorFilms} 的 {@code @JsonPropertyDescription} 里,改那里即可调整输出格式。
* curl -G "http://localhost:8081/api/chat/movies?model=openai&actor=%E6%A2%81%E6%9C%9D%E4%BC%9F"
* <br>或交由 curl 编码(UTF-8 终端下):
* curl -N -G "http://localhost:8081/api/chat/movies" --data-urlencode "actor=梁朝伟"
*
* <p>注意:推理型模型(如 deepseek-r1)的思考内容可能干扰 JSON 解析,报错时换非推理模型
* (如 qwen2.5)或走 openai 路由。
*/
@GetMapping("/movies")
public ActorFilms movies(@RequestParam(defaultValue = "ollama") String model,
@RequestParam(defaultValue = "梁朝伟") String actor) {
return ChatClient.create(resolve(model))
.prompt()
.user(u -> u.text("生成演员 {actor} 的 3~5 部代表作,"
+ "每部给出片名、上映年份、类型、导演、评分和一句话推荐理由。")
.param("actor", actor))
.call()
.entity(ActorFilms.class);
}
结构化输出
import com.fasterxml.jackson.annotation.JsonPropertyDescription;
import java.util.List;
/**
* 结构化输出用的目标类型(大纲阶段二)。
*
* <p>record 自动提供访问器,Spring AI 据此生成 JSON Schema 并塞进 Prompt,
* 模型按 Schema 输出 JSON 后自动反序列化。
*
* <p>关键点:{@code @JsonPropertyDescription} 会写进生成的 JSON Schema 的 description 字段,
* 这是告诉模型「每个字段要什么格式」的唯一手段 —— 描述越具体(尤其是单位、取值范围),
* 弱模型的表现越稳。改动这里的描述就等于改提示词,无需动 Controller。
*
* <p>嵌套 record 会生成嵌套对象 Schema,因此可以直接表达「一个演员对应多部电影」这种一对多结构。
*/
public record ActorFilms(
@JsonPropertyDescription("演员姓名,与输入保持一致") String actor,
@JsonPropertyDescription("代表作列表,3~5 部,按上映年份升序排列") List<Movie> movies) {
public record Movie(
@JsonPropertyDescription("电影片名") String title,
@JsonPropertyDescription("上映年份,四位数字,如 1994") Integer year,
@JsonPropertyDescription("电影类型,如 剧情 / 爱情 / 战争 / 科幻") String genre,
@JsonPropertyDescription("导演姓名") String director,
@JsonPropertyDescription("豆瓣评分,0~10 的浮点数,保留一位小数") Double rating,
@JsonPropertyDescription("一句话推荐理由,25 字以内") String highlight) {
}
}
测试模板输入: http://localhost:8081/api/chat/template?model=ollama&composer=久石让

测试结构化输出: http://localhost:8081/api/chat/movies?model=ollama&actor=梁朝伟

多轮记忆对话(SSE)
// ==================== 多轮记忆对话(SSE) ====================
/**
* 多轮记忆流式对话(SSE):同一 conversationId 会自动带上历史上下文。
* 图片示例:message="我叫小明,记得我",conversationId="user-001"。
*
* <p>POST 版本:前端用 fetch + ReadableStream 请求,支持自定义 Header。
* curl -N -X POST "http://localhost:8081/api/chat/memory" \\
* -d "model=ollama" \\
* -d "conversationId=user-001" \\
* --data-urlencode "message=我叫小明,记得我"
*/
@PostMapping(value = "/memory", produces = MediaType.TEXT_EVENT_STREAM_VALUE + ";charset=UTF-8")
public Flux<String> memoryByPost(@RequestParam(defaultValue = "ollama") String model,
@RequestParam String message,
@RequestParam String conversationId) {
return doMemoryChat(model, message, conversationId);
}
/**
* 多轮记忆流式对话(SSE):GET 版本,方便直接用 EventSource 测试。
* curl -N -G "http://localhost:8081/api/chat/memory" --data-urlencode "message=我叫小明,记得我" -d "conversationId=user-001"
*/
@GetMapping(value = "/memory", produces = MediaType.TEXT_EVENT_STREAM_VALUE + ";charset=UTF-8")
public Flux<String> memoryByGet(@RequestParam(defaultValue = "ollama") String model,
@RequestParam String message,
@RequestParam String conversationId) {
return doMemoryChat(model, message, conversationId);
}
private Flux<String> doMemoryChat(String model, String message, String conversationId) {
return ChatClient.create(resolve(model))
.prompt()
.user(message)
// 【记忆核心】Advisor = 引擎(负责读写历史),param = 钥匙(指定会话 ID),两者缺一不可
.advisors(a -> a
// 挂载记忆拦截器:请求前把该会话历史拼进 prompt,响应后把本轮问答写回 ChatMemory
.advisors(MessageChatMemoryAdvisor.builder(chatMemory).build())
// 告知 Advisor 去哪个会话读写;不传会回退默认 ID,导致所有会话互相串台
.param(ChatMemory.CONVERSATION_ID, conversationId))
.stream()
.content()
.filter(StringUtils::hasText)
.concatWith(Flux.just("[DONE]"));
}
测试(前几次89轮隔着都能记录,后面出现上面刚告诉下一轮问就不知道的情况)


原因分析:
deepseek-r1:14b,这是一个推理模型。这类模型输出本身就有随机性,即使上下文完全一样,两次回答也可能不同。尤其是 14b 这种小尺寸蒸馏版.
工具调用
// ==================== Function Calling / 工具调用 ====================
/**
* 工具调用:模型自动判断是否需要调用 LightingTools 控制灯光。
* 示例:message="把客厅的灯调暗到 30%"
*
* <p>POST http://localhost:8081/api/chat/tool
* curl -X POST "http://localhost:8081/api/chat/tool" -d "model=openai" --data-urlencode "message=把客厅的灯调暗到 30%"
*/
@PostMapping("/tool")
public String tool(@RequestParam(defaultValue = "openai") String model,
@RequestParam String message) {
return ChatClient.create(resolve(model))
.prompt()
.user(message)
.tools(lightingTools)
.call()
.content();
}
工具类
import java.util.Map;
import java.util.concurrent.ConcurrentHashMap;
import org.springframework.ai.tool.annotation.Tool;
import org.springframework.ai.tool.annotation.ToolParam;
import org.springframework.stereotype.Component;
/**
* 智能家居灯光控制工具(大纲阶段五:Function Calling)。
*
* <p>被 {@code @Tool} 标注的方法会被 Spring AI 自动注册为可调用的函数,
* 模型根据用户意图决定是否调用、并自动填充参数 —— 一个类里可以写多个工具方法,
* 全部会被扫描到,Controller 端无需逐个声明。
*
* <p>这里用一个内存 Map 模拟设备状态,方便看到「开灯 → 调暗 → 关灯」的真实效果;
* 生产环境换成智能家居 SDK / MQTT / Home Assistant 调用即可。
*/
@Component
public class LightingTools {
/** 房间 → 当前状态(on / brightness),内存模拟,重启丢失 */
private final Map<String, LightState> states = new ConcurrentHashMap<>();
/**
* 工具 1:开关灯。
*
* @param room 房间名称,如 客厅、卧室
* @param on true=开灯,false=关灯
* @return 操作结果描述
*/
@Tool(name = "switch_light", description = "打开或关闭指定房间的灯")
public String switchLight(
@ToolParam(required = true, description = "房间名称,例如 客厅、卧室") String room,
@ToolParam(required = true, description = "true 表示开灯,false 表示关灯") boolean on) {
LightState state = states.computeIfAbsent(room, r -> new LightState());
state.on = on;
System.out.println("[LightingTools] " + room + " 的灯已" + (on ? "打开" : "关闭"));
return "已把" + room + "的灯" + (on ? "打开" : "关闭");
}
/**
* 工具 2:调亮度。
*
* @param room 房间名称,如 客厅、卧室
* @param brightness 亮度百分比,0~100 的整数
* @return 操作结果描述
*/
@Tool(name = "adjust_light_brightness", description = "调整指定房间的灯光亮度")
public String adjustBrightness(
@ToolParam(required = true, description = "房间名称,例如 客厅、卧室") String room,
@ToolParam(required = true, description = "亮度百分比,0~100 的整数") int brightness) {
// 参数校验
if (brightness < 0 || brightness > 100) {
return "亮度必须在 0~100 之间,当前传入:" + brightness;
}
LightState state = states.computeIfAbsent(room, r -> new LightState());
// 调亮度隐含开灯
state.on = true;
state.brightness = brightness;
System.out.println("[LightingTools] 已将 " + room + " 的灯光亮度调整为 " + brightness + "%");
return "已把" + room + "的灯调到 " + brightness + "%";
}
/**
* 工具 3:查询当前灯光状态(可选,方便模型回答「客厅灯现在亮着吗」这类问题)。
*
* @param room 房间名称,如 客厅、卧室
* @return 该房间灯光的当前状态
*/
@Tool(name = "query_light_status", description = "查询指定房间灯光的开关状态和亮度")
public String queryStatus(
@ToolParam(required = true, description = "房间名称,例如 客厅、卧室") String room) {
LightState state = states.get(room);
if (state == null) {
return room + " 暂无灯光设备";
}
return state.on ? room + "的灯是开着的,亮度 " + state.brightness + "%" : room + "的灯是关着的";
}
/** 单个房间的灯光状态 */
private static class LightState {
private boolean on = false;
private int brightness = 100;
}
}
测试: curl.exe -X POST "http://localhost:8081/api/chat/tool" -d "model=ollama" --data-urlencode "message=把客厅的灯打开"

deepseek-r1:14b 是推理模型不能使用 工具调用
| 模型 | 工具支持 |
|---|---|
| qwen2.5 / qwen2.5:14b | 很好 |
| llama3.1 / llama3.1:8b | 官方支持 |
| mistral / mistral:7b | 支持 |
| phi4 | 支持 |
| deepseek-r1:14b | 不支持 / 极不稳定 |
换成qwen2.5调用成功

RAG检索配置
pom.xml添加依赖
<!-- RAG 向量库支持:VectorStore + SimpleVectorStore -->
<dependency>
<groupId>org.springframework.ai</groupId>
<artifactId>spring-ai-vector-store</artifactId>
</dependency>
<!-- RAG 检索 Advisor:QuestionAnswerAdvisor(独立于 vector-store,需单独引入) -->
<dependency>
<groupId>org.springframework.ai</groupId>
<artifactId>spring-ai-advisors-vector-store</artifactId>
</dependency>
<!-- RAG 文档读取:Tika 解析 PDF / Word / Excel / PPT / TXT 等本地文件 -->
<dependency>
<groupId>org.springframework.ai</groupId>
<artifactId>spring-ai-tika-document-reader</artifactId>
</dependency>
配置文件
application:
name: SpringAI-chat
ai:
ollama:
base-url: http://192.168.0.152:11434
chat:
options:
model: qwen2.5:14b #deepseek-r1:14b
temperature: 0.7
# RAG 向量库用的 Embedding 模型,首次使用需先执行:ollama pull nomic-embed-text
embedding:
options:
model: nomic-embed-text
# 自定义配置放最后:RAG 知识库文档目录(支持 PDF / Word / Excel / PPT / TXT)
# 把公司文档丢进该目录,启动时自动解析并向量化;目录为空则回退内置示例
rag:
documents-path: ./docs
RagConfig向量配置
import java.io.IOException;
import java.nio.file.Files;
import java.nio.file.Path;
import java.nio.file.Paths;
import java.util.ArrayList;
import java.util.List;
import java.util.stream.Stream;
import org.springframework.ai.document.Document;
import org.springframework.ai.ollama.OllamaEmbeddingModel;
import org.springframework.ai.reader.tika.TikaDocumentReader;
import org.springframework.ai.transformer.splitter.TokenTextSplitter;
import org.springframework.ai.vectorstore.SimpleVectorStore;
import org.springframework.ai.vectorstore.VectorStore;
import org.springframework.beans.factory.annotation.Value;
import org.springframework.context.annotation.Bean;
import org.springframework.context.annotation.Configuration;
import org.springframework.core.io.FileSystemResource;
/**
* RAG 检索增强配置(大纲阶段四):读取本地公司文档 → 切块 → 写入向量库。
*
* <p>文档目录由 {@code rag.documents-path} 指定,默认 {@code ./docs}。
* 把 PDF / Word / Excel / PPT / TXT 等文件直接丢进该目录即可,应用启动时自动加载。
*
* <p>解析用 Apache Tika({@link TikaDocumentReader}),能自动识别 1000+ 种格式,
* 无需为每种后缀写不同代码。若只需要读 PDF,可换成轻量得多的 {@code PagePdfDocumentReader}。
*
* <p>长文档必须先用 {@link TokenTextSplitter} 切块:整篇塞进 prompt 会超出模型上下文窗口,
* 且检索粒度太粗(一大段里只有一句相关也会整段召回),切块后检索精度才高。
*
* <p>目录为空时回退到内置示例文档,保证 /api/chat/rag/ask 始终可演示。
*/
@Configuration
public class RagConfig {
/**
* 本地公司文档目录。支持绝对路径(如 D:/company-docs)或相对项目根目录的路径。
*/
@Value("${rag.documents-path:./docs}")
private String documentsPath;
@Bean
public VectorStore vectorStore(OllamaEmbeddingModel ollamaEmbeddingModel) {
SimpleVectorStore vectorStore = SimpleVectorStore.builder(ollamaEmbeddingModel).build();
List<Document> chunks = loadFromDocuments();
if (chunks.isEmpty()) {
System.out.println("[RagConfig] 未读到本地文档,回退到内置示例知识库");
chunks = defaultDocuments();
}
vectorStore.add(chunks);
System.out.println("[RagConfig] 知识库加载完成,共 " + chunks.size() + " 个文本块");
return vectorStore;
}
/**
* 遍历文档目录,逐个用 Tika 解析,再统一切块。
*/
private List<Document> loadFromDocuments() {
Path dir = Paths.get(documentsPath);
if (!Files.isDirectory(dir)) {
System.out.println("[RagConfig] 文档目录不存在,跳过: " + dir.toAbsolutePath());
return List.of();
}
List<Document> raw = new ArrayList<>();
try (Stream<Path> files = Files.walk(dir)) {
List<Path> fileList = files
.filter(Files::isRegularFile)
// 跳过隐藏文件(如 macOS 的 .DS_Store)
.filter(p -> !p.getFileName().toString().startsWith("."))
.toList();
for (Path file : fileList) {
try {
TikaDocumentReader reader = new TikaDocumentReader(new FileSystemResource(file));
List<Document> docs = reader.get();
raw.addAll(docs);
System.out.println("[RagConfig] 已读取: " + file.getFileName() + " -> " + docs.size() + " 个文档");
} catch (Exception e) {
// 单个文件解析失败不影响其他文件
System.out.println("[RagConfig] 读取失败已跳过: " + file.getFileName() + " -> " + e.getMessage());
}
}
} catch (IOException e) {
System.out.println("[RagConfig] 遍历目录失败: " + e.getMessage());
return List.of();
}
if (raw.isEmpty()) {
return List.of();
}
// 切块:默认约 800 token/块,相邻块有重叠以保留上下文连贯性
return new TokenTextSplitter().apply(raw);
}
/**
* 内置示例知识库:docs 目录为空时使用,仅用于演示,接了真实文档后自然会走 {@link #loadFromDocuments()}。
*/
private List<Document> defaultDocuments() {
return List.of(
new Document("公司年假政策:员工入职满一年后,每年可享受 10 天带薪年假。"),
new Document("公司年假政策:入职不满一年的员工,按实际工作月份折算年假天数,每月 0.83 天。")
);
}
}

测试

Agent Chain 链式编排
// ==================== Agent Chain 链式编排 ====================
/**
* 两步链式编排:先出要点,再据要点扩写成文(大纲阶段六)。
*
* <p>这是最基础、也最实用的「工作流 / Agent 链」形态 —— 把一个复杂任务拆成两次
* 顺序调用,<b>上一步的输出作为下一步的输入</b>,而不是让模型一步到位:
* <ol>
* <li><b>第一步(出要点)</b>:让模型先做「策划」,只输出主题下的 N 条要点大纲,
* 任务单一、约束明确,弱模型也能稳定完成;</li>
* <li><b>第二步(扩写)</b>:把第一步得到的要点原样喂回,让模型「照着提纲写」,
* 把碎片要点扩写成结构完整、语言流畅的文章。</li>
* </ol>
*
* <p>为什么拆成两步,而不是一个 Prompt 全包?—— 拆解后每一步的 Prompt 更聚焦,
* 模型所需的「注意力」更小,输出质量与稳定性都更好;也便于在中间步骤插入
* 校验、改写、检索(RAG)等增强逻辑,这正是 Agent / 工作流编排的核心思路。
*
* <p>图片示例:topic="量子计算"。
*
* <p>一个入口同时支持 GET 与 POST(见下面 {@code method} 属性):
* <ul>
* <li><b>GET</b> —— 参数都有默认值,浏览器地址栏直接打开就能看效果:
* <pre>http://localhost:8081/api/agent/chain</pre>
* curl -G "http://localhost:8081/api/agent/chain" --data-urlencode "topic=量子计算"</li>
* <li><b>POST</b> —— 前端表单 / fetch 调用:
* curl -X POST "http://localhost:8081/api/agent/chain" -d "model=ollama" --data-urlencode "topic=量子计算"</li>
* </ul>
*
* <p>注意:URL 中的中文需百分号编码,否则部分浏览器/代理可能返回 400;
* 直接把中文粘进地址栏时浏览器会自动编码,一般无需手动处理。
*
* <p>返回 {@link ChainResult}(JSON):同时带出中间产物 {@code outline}(要点)与最终
* {@code article}(扩写),便于前端分段展示,也能直观看到「链」上每一步做了什么。
* 浏览器里会挤成一整行,装了 JSON 格式化插件或换成 {@code curl | jq} 更好读。
*
* <p>为什么写成 {@code @RequestMapping(method = {GET, POST})} 而不是
* {@code @GetMapping + @PostMapping} 两个方法:语义上是同一个「链」入口、逻辑完全一致,
* 合并后只有一份签名与一份 Javadoc,避免改一处漏一处。
*/
@RequestMapping(value = "/chain", method = {RequestMethod.GET, RequestMethod.POST})
public ChainResult chain(@RequestParam(defaultValue = "ollama") String model,
@RequestParam(defaultValue = "量子计算") String topic) {
return doChain(model, topic);
}
private ChainResult doChain(String model, String topic) {
ChatClient client = ChatClient.create(resolve(model));
// ---- 第一步:出要点(把「策划」这一步单独交给模型,任务聚焦、输出稳定)----
String outline = client.prompt()
.user(u -> u.text("你是内容策划。请针对主题「{topic}」列出 5 条要点大纲,"
+ "每条一句话,直接输出编号列表,不要额外的开场白或解释。")
.param("topic", topic))
.call()
.content();
// ---- 第二步:把上一步的要点作为输入,扩写成文(链式编排的关键:串起两次调用)----
String article = client.prompt()
.user(u -> u.text("请根据以下要点,扩写成一篇 300 字左右、结构完整、语言流畅的短文。"
+ "主题是「{topic}」,要点只是骨架,请自行补充衔接与过渡,不要只是罗列要点。\n\n"
+ "【要点】\n{outline}")
.param("topic", topic)
.param("outline", outline))
.call()
.content();
return new ChainResult(topic, outline, article);
}
/**
* Agent 链式编排的返回体:暴露中间产物与最终结果,方便前端分步展示。
*
* @param topic 原始输入主题
* @param outline 第一步产出的要点大纲(中间产物)
* @param article 第二步据要点扩写出的正文(最终结果)
*/
public record ChainResult(String topic, String outline, String article) {
}
测试

如果这篇文章对你有用,可以关注本人微信公众号获取更多ヽ(^ω^)ノ ~


浙公网安备 33010602011771号