Files
jbergner a6bc71fb3a
release-tag / Resolve release metadata (push) Successful in 30s
release-tag / Build knowledge (push) Failing after 4m51s
release-tag / Build control (push) Failing after 5m0s
release-tag / Build agent (push) Failing after 5m0s
release-tag / Build agent-data-init (push) Failing after 5m5s
release-tag / Build neuroforge-worker (push) Failing after 5m7s
release-tag / Build neuroforge (push) Failing after 5m9s
Init
2026-08-26 18:34:41 +02:00

243 lines
6.5 KiB
Go

package ingest
import (
"archive/zip"
"bytes"
"context"
"encoding/json"
"encoding/xml"
"errors"
"fmt"
"html"
"io"
"mime"
"os/exec"
"path/filepath"
"regexp"
"strings"
"unicode"
)
var (
reScript = regexp.MustCompile(`(?is)<script\b[^>]*>.*?</script>`)
reStyle = regexp.MustCompile(`(?is)<style\b[^>]*>.*?</style>`)
reComment = regexp.MustCompile(`(?is)<!--.*?-->`)
reTags = regexp.MustCompile(`(?s)<[^>]+>`)
reSpace = regexp.MustCompile(`[ \t\x0b\f\r]+`)
reBlank = regexp.MustCompile(`\n{3,}`)
)
// ExtractText extracts useful plain text from common knowledge-document formats.
// PDF support uses the optional pdftotext executable when it is installed; all
// other supported formats use only the Go standard library.
func ExtractText(name, contentType string, data []byte) (string, string, error) {
return ExtractTextContext(context.Background(), name, contentType, data)
}
func ExtractTextContext(ctx context.Context, name, contentType string, data []byte) (string, string, error) {
ext := strings.ToLower(filepath.Ext(name))
ct, _, _ := mime.ParseMediaType(contentType)
if ct == "" {
ct = contentType
}
switch {
case ext == ".html" || ext == ".htm" || ct == "text/html" || ct == "application/xhtml+xml":
return HTMLToText(string(data)), "text/html", nil
case ext == ".json" || ct == "application/json":
var v any
if json.Unmarshal(data, &v) == nil {
b, _ := json.MarshalIndent(v, "", " ")
return cleanText(string(b)), "application/json", nil
}
return cleanText(string(data)), "application/json", nil
case ext == ".txt" || ext == ".md" || ext == ".markdown" || ext == ".log" || ext == ".yaml" || ext == ".yml" || ext == ".csv" || ext == ".tsv" || strings.HasPrefix(ct, "text/"):
return cleanText(string(data)), nonempty(ct, "text/plain"), nil
case ext == ".docx" || ct == "application/vnd.openxmlformats-officedocument.wordprocessingml.document":
t, err := extractDOCX(data)
return t, "application/vnd.openxmlformats-officedocument.wordprocessingml.document", err
case ext == ".pdf" || ct == "application/pdf":
t, err := extractPDF(ctx, data)
return t, "application/pdf", err
default:
return "", ct, fmt.Errorf("unsupported document type %q (supported: txt, md, html, json, csv/tsv, docx, pdf with pdftotext)", ext)
}
}
func HTMLToText(in string) string {
s := reScript.ReplaceAllString(in, " ")
s = reStyle.ReplaceAllString(s, " ")
s = reComment.ReplaceAllString(s, " ")
// preserve rough block boundaries before removing tags.
r := strings.NewReplacer("</p>", "\n\n", "</div>", "\n", "</li>", "\n", "<br>", "\n", "<br/>", "\n", "<br />", "\n", "</h1>", "\n\n", "</h2>", "\n\n", "</h3>", "\n\n")
s = r.Replace(s)
s = reTags.ReplaceAllString(s, " ")
s = html.UnescapeString(s)
return cleanText(s)
}
func cleanText(s string) string {
s = strings.ReplaceAll(s, "\x00", "")
lines := strings.Split(s, "\n")
out := make([]string, 0, len(lines))
for _, line := range lines {
line = strings.TrimSpace(reSpace.ReplaceAllString(line, " "))
if line != "" {
out = append(out, line)
} else if len(out) > 0 && out[len(out)-1] != "" {
out = append(out, "")
}
}
return strings.TrimSpace(reBlank.ReplaceAllString(strings.Join(out, "\n"), "\n\n"))
}
func extractDOCX(data []byte) (string, error) {
zr, err := zip.NewReader(bytes.NewReader(data), int64(len(data)))
if err != nil {
return "", fmt.Errorf("open docx: %w", err)
}
var doc *zip.File
for _, f := range zr.File {
if f.Name == "word/document.xml" {
doc = f
break
}
}
if doc == nil {
return "", errors.New("docx has no word/document.xml")
}
rc, err := doc.Open()
if err != nil {
return "", err
}
defer rc.Close()
dec := xml.NewDecoder(io.LimitReader(rc, 64<<20))
var b strings.Builder
for {
tok, err := dec.Token()
if err == io.EOF {
break
}
if err != nil {
return "", err
}
switch t := tok.(type) {
case xml.StartElement:
if t.Name.Local == "t" {
var text string
if err := dec.DecodeElement(&text, &t); err != nil {
return "", err
}
b.WriteString(text)
}
case xml.EndElement:
if t.Name.Local == "p" {
b.WriteString("\n\n")
} else if t.Name.Local == "tab" {
b.WriteByte('\t')
}
}
}
return cleanText(b.String()), nil
}
func extractPDF(ctx context.Context, data []byte) (string, error) {
path, err := exec.LookPath("pdftotext")
if err != nil {
return "", errors.New("PDF extraction requires the optional 'pdftotext' executable (poppler-utils); install it or convert the PDF to text/HTML first")
}
cmd := exec.CommandContext(ctx, path, "-layout", "-", "-")
cmd.Stdin = bytes.NewReader(data)
var stderr bytes.Buffer
out := &cappedBuffer{limit: 64 << 20}
cmd.Stdout = out
cmd.Stderr = &stderr
if err := cmd.Run(); err != nil {
if errors.Is(out.err, errExtractedTextTooLarge) {
return "", errExtractedTextTooLarge
}
return "", fmt.Errorf("pdftotext: %v: %s", err, strings.TrimSpace(stderr.String()))
}
return cleanText(out.buf.String()), nil
}
var errExtractedTextTooLarge = errors.New("extracted document text exceeds 64 MiB safety limit")
type cappedBuffer struct {
buf bytes.Buffer
limit int
err error
}
func (w *cappedBuffer) Write(p []byte) (int, error) {
if w.err != nil {
return 0, w.err
}
if w.buf.Len()+len(p) > w.limit {
w.err = errExtractedTextTooLarge
return 0, w.err
}
return w.buf.Write(p)
}
func ChunkText(text string, chunkChars, overlap, maxChunks int) []string {
text = cleanText(text)
if text == "" {
return nil
}
if chunkChars <= 0 {
chunkChars = 2400
}
if overlap < 0 {
overlap = 0
}
if overlap >= chunkChars {
overlap = chunkChars / 8
}
if maxChunks <= 0 {
maxChunks = 2000
}
r := []rune(text)
chunks := make([]string, 0, min(maxChunks, len(r)/chunkChars+1))
for start := 0; start < len(r) && len(chunks) < maxChunks; {
end := start + chunkChars
if end >= len(r) {
end = len(r)
} else {
// Try to end at a paragraph/sentence/space boundary without shrinking too much.
floor := start + chunkChars*3/4
for i := end; i > floor; i-- {
if r[i-1] == '\n' || r[i-1] == '.' || r[i-1] == '!' || r[i-1] == '?' || unicode.IsSpace(r[i-1]) {
end = i
break
}
}
}
chunk := strings.TrimSpace(string(r[start:end]))
if chunk != "" {
chunks = append(chunks, chunk)
}
if end >= len(r) {
break
}
next := end - overlap
if next <= start {
next = end
}
start = next
}
return chunks
}
func nonempty(v, fallback string) string {
if strings.TrimSpace(v) == "" {
return fallback
}
return v
}
func min(a, b int) int {
if a < b {
return a
}
return b
}