fix(ui): 自定义工具表单 overflow-y-scroll→auto + 认证方式NONE翻译改为'无' + Captain文档爬虫/同步后端
This commit is contained in:
@@ -0,0 +1,175 @@
|
||||
package service
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"os"
|
||||
"os/exec"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/gochat/gochat/internal/model"
|
||||
"golang.org/x/net/html"
|
||||
)
|
||||
|
||||
// captainDocumentSyncBackendImpl is the production implementation of
|
||||
// CaptainDocumentSyncBackend. It handles two document types:
|
||||
// - PDF uploads: extracts text via the `pdftotext` CLI (poppler-utils),
|
||||
// which correctly handles CJK/CID fonts that pure-Go PDF libraries cannot.
|
||||
// - Web URLs: fetches the page and extracts visible text from HTML.
|
||||
type captainDocumentSyncBackendImpl struct {
|
||||
httpClient *http.Client
|
||||
}
|
||||
|
||||
// NewCaptainDocumentSyncBackend creates the production sync backend.
|
||||
func NewCaptainDocumentSyncBackend() CaptainDocumentSyncBackend {
|
||||
return &captainDocumentSyncBackendImpl{
|
||||
httpClient: &http.Client{Timeout: 30 * time.Second},
|
||||
}
|
||||
}
|
||||
|
||||
func (b *captainDocumentSyncBackendImpl) SyncCaptainDocument(ctx context.Context, doc *model.CaptainDocument) (*CaptainDocumentSyncResult, error) {
|
||||
// PDF document: extract text from the uploaded file.
|
||||
if doc.ContentType == "application/pdf" || doc.FileURL != "" {
|
||||
return b.syncPDFDocument(ctx, doc)
|
||||
}
|
||||
// Web URL document: fetch and extract text from the page.
|
||||
if doc.ExternalLink != "" {
|
||||
return b.syncWebDocument(ctx, doc)
|
||||
}
|
||||
return &CaptainDocumentSyncResult{ErrorCode: "content_empty"}, nil
|
||||
}
|
||||
|
||||
// syncPDFDocument extracts text content from a locally stored PDF file
|
||||
// using the `pdftotext` CLI (poppler-utils). This handles CJK fonts and
|
||||
// complex PDF encodings that pure-Go libraries like ledongthuc/pdf cannot.
|
||||
func (b *captainDocumentSyncBackendImpl) syncPDFDocument(ctx context.Context, doc *model.CaptainDocument) (*CaptainDocumentSyncResult, error) {
|
||||
filePath := b.resolvePDFPath(doc)
|
||||
if filePath == "" {
|
||||
return &CaptainDocumentSyncResult{ErrorCode: "not_found"}, nil
|
||||
}
|
||||
|
||||
// Use pdftotext CLI for robust text extraction (supports CJK).
|
||||
cmd := exec.CommandContext(ctx, "pdftotext", "-enc", "UTF-8", filePath, "-")
|
||||
var stdout, stderr bytes.Buffer
|
||||
cmd.Stdout = &stdout
|
||||
cmd.Stderr = &stderr
|
||||
if err := cmd.Run(); err != nil {
|
||||
return &CaptainDocumentSyncResult{ErrorCode: "fetch_failed"}, fmt.Errorf("pdftotext: %w, stderr: %s", err, stderr.String())
|
||||
}
|
||||
|
||||
content := strings.TrimSpace(stdout.String())
|
||||
if content == "" {
|
||||
return &CaptainDocumentSyncResult{ErrorCode: "content_empty"}, nil
|
||||
}
|
||||
|
||||
title := doc.Name
|
||||
return &CaptainDocumentSyncResult{
|
||||
Content: content,
|
||||
Title: title,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// resolvePDFPath converts the doc's file_url to a local filesystem path.
|
||||
// file_url is stored as "/uploads/captain_docs/<account>/<timestamp>.pdf"
|
||||
// and served from the local "./uploads" directory.
|
||||
func (b *captainDocumentSyncBackendImpl) resolvePDFPath(doc *model.CaptainDocument) string {
|
||||
if doc.FileURL == "" {
|
||||
return ""
|
||||
}
|
||||
// file_url is a relative URL path like "/uploads/captain_docs/1/123.pdf"
|
||||
// Map it to the local filesystem path.
|
||||
path := doc.FileURL
|
||||
if strings.HasPrefix(path, "/uploads/") {
|
||||
return "." + path
|
||||
}
|
||||
// If it's already a filesystem path, use it directly.
|
||||
if _, err := os.Stat(path); err == nil {
|
||||
return path
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// syncWebDocument fetches a web page and extracts its visible text content.
|
||||
func (b *captainDocumentSyncBackendImpl) syncWebDocument(ctx context.Context, doc *model.CaptainDocument) (*CaptainDocumentSyncResult, error) {
|
||||
url := doc.ExternalLink
|
||||
if url == "" {
|
||||
return &CaptainDocumentSyncResult{ErrorCode: "not_found"}, nil
|
||||
}
|
||||
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)
|
||||
if err != nil {
|
||||
return &CaptainDocumentSyncResult{ErrorCode: "fetch_failed"}, fmt.Errorf("create request: %w", err)
|
||||
}
|
||||
req.Header.Set("User-Agent", "GoChat-Captain/1.0")
|
||||
|
||||
resp, err := b.httpClient.Do(req)
|
||||
if err != nil {
|
||||
return &CaptainDocumentSyncResult{ErrorCode: "fetch_failed"}, fmt.Errorf("fetch url: %w", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return &CaptainDocumentSyncResult{ErrorCode: "fetch_failed"}, fmt.Errorf("HTTP %d for %s", resp.StatusCode, url)
|
||||
}
|
||||
|
||||
body, err := io.ReadAll(io.LimitReader(resp.Body, 5<<20)) // 5MB max
|
||||
if err != nil {
|
||||
return &CaptainDocumentSyncResult{ErrorCode: "fetch_failed"}, fmt.Errorf("read body: %w", err)
|
||||
}
|
||||
|
||||
title, content := extractHTMLText(string(body))
|
||||
content = strings.TrimSpace(content)
|
||||
if content == "" {
|
||||
return &CaptainDocumentSyncResult{ErrorCode: "content_empty"}, nil
|
||||
}
|
||||
|
||||
if title == "" {
|
||||
title = doc.Name
|
||||
}
|
||||
return &CaptainDocumentSyncResult{
|
||||
Content: content,
|
||||
Title: title,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// extractHTMLText parses an HTML document and returns the page title and
|
||||
// visible text content (stripping scripts, styles, and HTML tags).
|
||||
func extractHTMLText(htmlStr string) (title string, content string) {
|
||||
doc, err := html.Parse(strings.NewReader(htmlStr))
|
||||
if err != nil {
|
||||
// Fall back to raw text if HTML parsing fails.
|
||||
return "", strings.TrimSpace(htmlStr)
|
||||
}
|
||||
|
||||
var extract func(*html.Node)
|
||||
extract = func(n *html.Node) {
|
||||
if n.Type == html.ElementNode {
|
||||
switch n.Data {
|
||||
case "script", "style", "noscript", "head":
|
||||
return
|
||||
case "title":
|
||||
if n.FirstChild != nil {
|
||||
title = strings.TrimSpace(n.FirstChild.Data)
|
||||
}
|
||||
return
|
||||
}
|
||||
}
|
||||
if n.Type == html.TextNode {
|
||||
text := strings.TrimSpace(n.Data)
|
||||
if text != "" {
|
||||
content += text + " "
|
||||
}
|
||||
}
|
||||
for c := n.FirstChild; c != nil; c = c.NextSibling {
|
||||
extract(c)
|
||||
}
|
||||
}
|
||||
extract(doc)
|
||||
|
||||
content = strings.TrimSpace(content)
|
||||
return title, content
|
||||
}
|
||||
Reference in New Issue
Block a user