diff --git a/workflow/service/flow/lambda_node_imp.go b/workflow/service/flow/lambda_node_imp.go index 8019385..972310f 100644 --- a/workflow/service/flow/lambda_node_imp.go +++ b/workflow/service/flow/lambda_node_imp.go @@ -683,24 +683,29 @@ func AudioOptimizeNode(ctx context.Context, nodeInput *flowDto.NodeExecutionInpu return outputRes, nil } +// splitTextByPunct 按中文标点分割句子,同时保留标点在分段内 +// 例如:"这个叫高血压调理方,注意是根源调理不是临时缓解," +// 会变成:["这个叫高血压调理方,", "注意是根源调理不是临时缓解,"] func splitTextByPunct(raw string) []string { - // 按标点切分+拼接标点 - slice := regexp.MustCompile(`([,。;!?])`).Split(raw, -1) - var res []string - var builder strings.Builder - for idx, s := range slice { - if s == "" { - continue - } - builder.WriteString(s) - // 偶数位是分隔标点(split后规律:文本、标点、文本、标点...) - if idx%2 == 1 { - res = append(res, builder.String()) - builder.Reset() - } + // 匹配中文标点并保留在文本中,按标点位置切分 + re := regexp.MustCompile(`[,。;!?]`) + // 先找到所有标点的位置 + indexes := re.FindAllStringIndex(raw, -1) + if len(indexes) == 0 { + return []string{raw} } - if builder.Len() > 0 { - res = append(res, builder.String()) + + var res []string + prev := 0 + for _, idx := range indexes { + end := idx[1] // 标点的结束位置 + seg := raw[prev:end] + res = append(res, seg) + prev = end + } + // 处理最后一段没有标点的文本 + if prev < len(raw) { + res = append(res, raw[prev:]) } return res } @@ -708,46 +713,53 @@ func splitTextByPunct(raw string) []string { // BuildSubtitles 核心工具:单个sentence生成多条subtitle func BuildSubtitles(sents *[]flowDto.Sentence) ([]flowDto.Subtitle, error) { var subtitles []flowDto.Subtitle + for _, sent := range *sents { + // 1. 先按标点把文本拆成多个片段(保留标点) segList := splitTextByPunct(sent.Text) if len(segList) == 0 { - return nil, nil + continue } - var subs []flowDto.Subtitle wordIdx := 0 allWords := sent.Words - + // 2. 遍历每个文本片段,匹配对应的Words for _, seg := range segList { + // 去除文本片段的标点,方便和Word.Word拼接内容匹配 + segClean := strings.ReplaceAll(seg, ",", "") + segClean = strings.ReplaceAll(segClean, "。", "") + segClean = strings.ReplaceAll(segClean, ";", "") + segClean = strings.ReplaceAll(segClean, "!", "") + segClean = strings.ReplaceAll(segClean, "?", "") + var collectWords []flowDto.Word - currentText := "" - // 循环取 word,直到拼接内容 包含/匹配 seg - for { - if wordIdx >= len(allWords) { - break - } + var currentText strings.Builder + + // 收集Word直到拼接内容覆盖当前分段 + for wordIdx < len(allWords) { word := allWords[wordIdx] - currentText += word.Word + currentText.WriteString(word.Word) collectWords = append(collectWords, word) wordIdx++ - // 只要包含分段文本,就认为匹配(无视末尾标点差异) - if strings.Contains(currentText, seg) { + // 当拼接的文本包含当前分段的纯文本时,停止收集 + if strings.Contains(currentText.String(), segClean) { break } } + if len(collectWords) == 0 { continue } - // 生成字幕 + + // 3. 生成字幕(时间戳取首尾Word的时间) sub := flowDto.Subtitle{ Start: collectWords[0].StartTime, End: collectWords[len(collectWords)-1].EndTime, - Text: seg, + Text: segClean, } - subs = append(subs, sub) + subtitles = append(subtitles, sub) } - subtitles = append(subtitles, subs...) } return subtitles, nil @@ -792,12 +804,12 @@ func DataConversionNode(ctx context.Context, nodeInput *flowDto.NodeExecutionInp } jsonStr := `` - jsonVal := "输出字段规范:" + jsonVal := "" for _, field := range nodeInput.Config.OutputConfig { jsonStr, _ = sjson.Set(jsonStr, field.Field, "") - jsonVal += fmt.Sprintf("%s:%s;", field.Field, field.Value) + //jsonVal += fmt.Sprintf("%s:%s;", field.Field, field.Value) } - jsonVal += fmt.Sprintf("输出模板结构,仅修改每个字段对应数值:%s", jsonStr) + jsonVal += fmt.Sprintf("输出字段规范:%v", jsonStr) nodeInput.Config.PromptContent = fmt.Sprintf("%s;%s", nodeInput.Config.PromptContent, jsonVal) mapTaskResult, err := GetModelResult(ctx, "", nodeInput, skillName, form, userForm) if err != nil {