You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
198 lines
7.9 KiB
198 lines
7.9 KiB
package memory
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"fmt"
|
|
"strings"
|
|
"time"
|
|
|
|
"yunyan/comm"
|
|
"yunyan/pb"
|
|
)
|
|
|
|
/*
|
|
会议纪要 → 待办抽取(第三路)。
|
|
|
|
在 echomeet 的 AIProcess 写完 summary/overview、状态置 Completed **之后**由它回调进来,
|
|
独立跑、失败只记日志不改 rec.State —— 用户照样看得到纪要,只是没有自动待办。
|
|
|
|
几个刻意的选择:
|
|
|
|
- **prompt 放代码不放 echomeet_template**:模板有 1189 条(29 组 × 41 语言),改不动也改不齐;
|
|
各模板 outline 职责本就不同,塞进去会互相干扰;输出要被程序 parse 成 JSON,
|
|
格式严格性不能交给后台可随手编辑的文本。
|
|
- **输入用已生成的 summary 而不是转写全文**:summary 约千字,全文可达上万字,
|
|
成本只是前两路的零头。
|
|
- **必须注入会议日期**:不给日期,模型把「下周三前」换算成具体日期就是瞎猜。
|
|
- **复用 rec.LlmSvcId**:与本条会议的总结同一个模型。同一条会议两路用不同模型,
|
|
出问题时无法归因。
|
|
*/
|
|
|
|
const meetingExtractPrompt = `你是一个会议待办抽取器。用户会给你一份会议纪要,以及这次会议的日期。
|
|
请从纪要的「待办事项」部分抽取出结构化的任务列表。
|
|
|
|
严格要求:
|
|
1. 只输出一个 JSON 数组,不要 markdown 代码块,不要任何解释文字。数组为空就输出 []。
|
|
2. 数组每一项的字段固定为:
|
|
{"title":"任务内容","owner":"负责人","due_date":"YYYY-MM-DD","due_raw":"截止时间原文"}
|
|
3. title 必填。owner / due_date / due_raw 没有就填空字符串 ""。
|
|
4. **title 与 owner 必须与输入纪要使用同一种语言**,不要翻译成中文。
|
|
5. due_date 只在能从原文明确推算出具体日期时才填,推算基准是给定的会议日期。
|
|
含糊的表述(「尽快」「月底前」无法定位到具体某天)一律留空 due_date,
|
|
把原文放进 due_raw。**禁止猜测、禁止编造日期。**
|
|
6. 只收录会议中明确要求执行的任务;仅在讨论中提及、未拍板的设想不计入。
|
|
7. 同一件事被多次提及只输出一条,取最终确定的版本。`
|
|
|
|
// extractedTodo 抽取结果的一条
|
|
type extractedTodo struct {
|
|
Title string `json:"title"`
|
|
Owner string `json:"owner"`
|
|
DueDate string `json:"due_date"`
|
|
DueRaw string `json:"due_raw"`
|
|
}
|
|
|
|
// 一次会议最多落多少条待办。模型偶尔会把整篇纪要拆成几十条,
|
|
// 全落下去会把用户的日历那一天彻底刷爆。
|
|
const maxExtractPerMeeting = 30
|
|
|
|
// ExtractMeetingTodos 从会议纪要抽待办并落库(实现 comm.IMemory)。
|
|
//
|
|
// meetingDate 是会议归属日期(YYYY-MM-DD);抽不出截止日期的项落在这一天并置
|
|
// date_certain=false,客户端应把它们归到「待定日期」分组,不要和当天确定事项混排
|
|
// —— 一次会抽出 8 条没写截止时间的待办全堆在会议当天,那一格就没法看了。
|
|
func (this *Memory) ExtractMeetingTodos(ctx context.Context, uid, recordID, summary, meetingDate, llmSvcID string) {
|
|
defer func() {
|
|
if r := recover(); r != nil {
|
|
this.Errorf("会议待办抽取 panic record:%s err:%v", recordID, r)
|
|
}
|
|
}()
|
|
if !this.options.MeetingExtract {
|
|
return
|
|
}
|
|
if strings.TrimSpace(summary) == "" || uid == "" {
|
|
return
|
|
}
|
|
if _, ok := comm.ParseMemoryDate(meetingDate); !ok {
|
|
this.Warnf("会议待办抽取 record:%s 会议日期非法(%q),跳过", recordID, meetingDate)
|
|
return
|
|
}
|
|
if this.echo == nil {
|
|
this.Warnf("会议待办抽取 record:%s 未取到 echomeet,跳过", recordID)
|
|
return
|
|
}
|
|
|
|
// 本次是第几轮:取这条会议现有项的最大 gen_round + 1。
|
|
// 清理时删的是 gen_round < 本轮 且 user_edited=false 的项,所以轮次必须真的递增——
|
|
// 恒为 1 的话第二次重新生成会一条都清不掉,旧待办和新待办并存。
|
|
genRound := this.model.nextGenRound(recordID)
|
|
|
|
prompt := this.options.MeetingExtractPrompt
|
|
if prompt == "" {
|
|
prompt = meetingExtractPrompt
|
|
}
|
|
userContent := fmt.Sprintf("<meeting_date>%s</meeting_date>\n\n<summary>\n%s\n</summary>", meetingDate, summary)
|
|
|
|
c, cancel := context.WithTimeout(ctx, 90*time.Second)
|
|
text, usedSvc, err := this.echo.ChatLLM(c, llmSvcID, prompt, userContent)
|
|
cancel()
|
|
if err != nil {
|
|
// 失败隔离:只记日志,绝不改 rec.State。纪要是用户真正要的东西,
|
|
// 不能因为附赠的待办抽取失败就把它标成失败。
|
|
this.Errorf("会议待办抽取 record:%s 调用失败 svc:%s err:%v", recordID, usedSvc, err)
|
|
return
|
|
}
|
|
|
|
todos, err := parseExtractedTodos(text)
|
|
if err != nil {
|
|
this.Warnf("会议待办抽取 record:%s 解析失败已忽略 err:%v raw:%.200s", recordID, err, text)
|
|
return
|
|
}
|
|
if len(todos) == 0 {
|
|
this.Infof("会议待办抽取 record:%s 无待办", recordID)
|
|
return
|
|
}
|
|
if len(todos) > maxExtractPerMeeting {
|
|
this.Warnf("会议待办抽取 record:%s 返回 %d 条,截断到 %d", recordID, len(todos), maxExtractPerMeeting)
|
|
todos = todos[:maxExtractPerMeeting]
|
|
}
|
|
|
|
// 幂等:清掉上一轮**自动生成**的项,user_edited 的一律保留。
|
|
// 不做这个,用户点一次「重新生成」就会丢掉自己的修改。
|
|
if n, e := this.model.delMeetingItems(recordID, genRound); e != nil {
|
|
this.Warnf("会议待办抽取 record:%s 清理旧轮次失败已忽略 err:%v", recordID, e)
|
|
} else if n > 0 {
|
|
this.Infof("会议待办抽取 record:%s 清理旧轮次 %d 条", recordID, n)
|
|
}
|
|
|
|
meetDay, _ := comm.ParseMemoryDate(meetingDate)
|
|
saved := 0
|
|
for i, t := range todos {
|
|
t.Title = strings.TrimSpace(t.Title)
|
|
if t.Title == "" {
|
|
continue
|
|
}
|
|
item := &pb.DBMemoryItem{
|
|
Uid: uid, Category: comm.MemoryCatTodo, Source: comm.MemorySrcMeeting,
|
|
SourceId: recordID,
|
|
// client_key 用「记录id + 轮次 + 序号」,重跑同一轮不会插重
|
|
ClientKey: fmt.Sprintf("meet:%s:%d:%d", recordID, genRound, i),
|
|
Title: t.Title,
|
|
Owner: strings.TrimSpace(t.Owner),
|
|
DueRaw: strings.TrimSpace(t.DueRaw),
|
|
HappenDate: meetingDate,
|
|
DateCertain: false,
|
|
GenRound: genRound,
|
|
State: pb.MemoryState_MemoryState_Pending,
|
|
RemindAhead: 0, // 会议待办默认不提醒,用户改期时自己开
|
|
Currency: "CNY",
|
|
}
|
|
// 「未明确」是模板要求模型在原文没写时填的占位词,不是真的负责人
|
|
if item.Owner == "未明确" || strings.EqualFold(item.Owner, "unspecified") {
|
|
item.Owner = ""
|
|
}
|
|
|
|
// 服务端二次校验日期:解析不出、或早于会议当天的一律丢弃,只留 due_raw。
|
|
// 参考 SET_clock 踩过的坑——模型把「下午 3:30」的 time 填成 03:30。
|
|
if d, ok := comm.ParseMemoryDate(strings.TrimSpace(t.DueDate)); ok {
|
|
if !d.Before(meetDay) {
|
|
item.HappenDate = comm.FormatMemoryDate(d)
|
|
item.DateCertain = true
|
|
} else {
|
|
this.Warnf("会议待办抽取 record:%s 丢弃早于会议日的截止日 %s", recordID, t.DueDate)
|
|
}
|
|
}
|
|
|
|
if err := this.model.addItem(item); err != nil {
|
|
this.Warnf("会议待办抽取 record:%s 第%d条落库失败已跳过 err:%v", recordID, i, err)
|
|
continue
|
|
}
|
|
saved++
|
|
}
|
|
this.Infof("会议待办抽取 record:%s uid:%s svc:%s 落库 %d/%d 条", recordID, uid, usedSvc, saved, len(todos))
|
|
}
|
|
|
|
// parseExtractedTodos 解析模型输出。
|
|
//
|
|
// 模型时常无视「不要 markdown 代码块」这条,所以先剥一层 ```json ... ```;
|
|
// 也可能在数组前后带一句解释,再按第一个 [ 到最后一个 ] 截一次。
|
|
func parseExtractedTodos(text string) ([]extractedTodo, error) {
|
|
s := strings.TrimSpace(text)
|
|
if strings.HasPrefix(s, "```") {
|
|
if i := strings.Index(s, "\n"); i >= 0 {
|
|
s = s[i+1:]
|
|
}
|
|
if i := strings.LastIndex(s, "```"); i >= 0 {
|
|
s = s[:i]
|
|
}
|
|
s = strings.TrimSpace(s)
|
|
}
|
|
if start, end := strings.Index(s, "["), strings.LastIndex(s, "]"); start >= 0 && end > start {
|
|
s = s[start : end+1]
|
|
}
|
|
out := make([]extractedTodo, 0)
|
|
if err := json.Unmarshal([]byte(s), &out); err != nil {
|
|
return nil, err
|
|
}
|
|
return out, nil
|
|
}
|
|
|