pipeline.Run parses source/author/published from the raw content's own frontmatter and applies them deterministically to source-type RawPages after LLM extraction — never relies on the LLM to copy them through. buildFrontmatter now emits all three (source pages only) when present. Backfill of the 46 already-affected wiki/sources/*.md notes is a separate concern per the issue, not done here. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Roq1ajWKR5f1hG5Df9wC6A
173 lines
5.1 KiB
Go
173 lines
5.1 KiB
Go
// ingestion/internal/pipeline/parse.go
|
|
package pipeline
|
|
|
|
import (
|
|
"encoding/json"
|
|
"fmt"
|
|
"strings"
|
|
)
|
|
|
|
// RawPage is the LLM's output format — minimal structured data with no path or frontmatter.
|
|
// The pipeline derives slugs, paths, and frontmatter from these fields.
|
|
type RawPage struct {
|
|
Title string `json:"title"`
|
|
Type string `json:"type"` // "source" | "concept" | "entity"
|
|
Subtype string `json:"subtype"` // entity: person|company|tool|model|framework|technology; source: article|pdf|book|video|note|project
|
|
Domain string `json:"domain"`
|
|
Content string `json:"content"` // Markdown body only — no frontmatter
|
|
|
|
// Source, Author, Published are deterministic passthrough from the raw
|
|
// ingested content's own frontmatter (see parseContentFrontmatter) — never
|
|
// set by the LLM. json:"-" keeps them immune to same-named keys the LLM
|
|
// might emit. Only meaningful for Type == "source".
|
|
Source string `json:"-"`
|
|
Author string `json:"-"`
|
|
Published string `json:"-"`
|
|
}
|
|
|
|
// ParseRawPages parses LLM output as a JSON array of RawPage objects.
|
|
// If the output contains invalid JSON escape sequences (e.g. \. from Markdown),
|
|
// it attempts repair before falling back to truncation recovery.
|
|
func ParseRawPages(output string) ([]RawPage, []string) {
|
|
output = strings.TrimSpace(output)
|
|
if output == "" {
|
|
return nil, []string{"LLM returned empty output"}
|
|
}
|
|
|
|
output = stripFences(output)
|
|
|
|
// Fast path: valid JSON.
|
|
var pages []RawPage
|
|
if err := json.Unmarshal([]byte(output), &pages); err == nil {
|
|
return pages, nil
|
|
}
|
|
|
|
// Repair pass: fix invalid escape sequences (e.g. \. \d from Markdown content).
|
|
repaired := repairJSON(output)
|
|
if err := json.Unmarshal([]byte(repaired), &pages); err == nil {
|
|
return pages, []string{"repaired invalid JSON escape sequences in LLM output"}
|
|
}
|
|
|
|
// Truncation recovery: find last `}` that closes a complete object.
|
|
idx := strings.LastIndex(repaired, "}")
|
|
if idx < 0 {
|
|
return nil, []string{"LLM output contained no complete JSON objects"}
|
|
}
|
|
|
|
start := strings.Index(repaired, "[")
|
|
if start < 0 {
|
|
return nil, []string{"LLM output contained no JSON array opening bracket"}
|
|
}
|
|
|
|
candidate := repaired[start:idx+1] + "]"
|
|
if err := json.Unmarshal([]byte(candidate), &pages); err != nil {
|
|
return nil, []string{fmt.Sprintf("truncation recovery failed: %v", err)}
|
|
}
|
|
|
|
return pages, []string{fmt.Sprintf("LLM output was truncated; recovered %d page(s)", len(pages))}
|
|
}
|
|
|
|
// repairJSON replaces invalid JSON escape sequences (e.g. \. \d \p) with
|
|
// a properly escaped backslash followed by the same character.
|
|
// It iterates byte-by-byte to correctly skip already-valid escape sequences
|
|
// (including \\) without requiring lookbehind support.
|
|
func repairJSON(s string) string {
|
|
var b strings.Builder
|
|
b.Grow(len(s))
|
|
i := 0
|
|
for i < len(s) {
|
|
if s[i] != '\\' {
|
|
b.WriteByte(s[i])
|
|
i++
|
|
continue
|
|
}
|
|
// We have a backslash. Peek at the next character.
|
|
if i+1 >= len(s) {
|
|
// Trailing backslash — emit as-is.
|
|
b.WriteByte(s[i])
|
|
i++
|
|
continue
|
|
}
|
|
next := s[i+1]
|
|
switch next {
|
|
case '"', '\\', '/', 'b', 'f', 'n', 'r', 't', 'u':
|
|
// Valid JSON escape sequence — emit both characters as-is.
|
|
b.WriteByte(s[i])
|
|
b.WriteByte(next)
|
|
i += 2
|
|
default:
|
|
// Invalid escape — double the backslash.
|
|
b.WriteByte('\\')
|
|
b.WriteByte('\\')
|
|
b.WriteByte(next)
|
|
i += 2
|
|
}
|
|
}
|
|
return b.String()
|
|
}
|
|
|
|
// sourceMeta is source/author/published pulled from the raw ingested
|
|
// content's own frontmatter — deterministic passthrough, never LLM output.
|
|
type sourceMeta struct {
|
|
Source string
|
|
Author string
|
|
Published string
|
|
}
|
|
|
|
// parseContentFrontmatter extracts source/author/published from a leading
|
|
// "---\n...\n---" YAML block in raw ingested content. Only these three flat
|
|
// scalar keys are recognised; anything else in the block is ignored. Returns
|
|
// a zero-value sourceMeta if content has no frontmatter block.
|
|
func parseContentFrontmatter(content string) sourceMeta {
|
|
var meta sourceMeta
|
|
if !strings.HasPrefix(content, "---\n") && !strings.HasPrefix(content, "---\r\n") {
|
|
return meta
|
|
}
|
|
|
|
lines := strings.Split(content, "\n")
|
|
for _, line := range lines[1:] {
|
|
if strings.TrimSpace(line) == "---" {
|
|
break
|
|
}
|
|
key, val, ok := strings.Cut(line, ":")
|
|
if !ok {
|
|
continue
|
|
}
|
|
key = strings.TrimSpace(key)
|
|
val = strings.Trim(strings.TrimSpace(val), `"'`)
|
|
switch key {
|
|
case "source":
|
|
meta.Source = val
|
|
case "author":
|
|
meta.Author = val
|
|
case "published":
|
|
meta.Published = val
|
|
}
|
|
}
|
|
return meta
|
|
}
|
|
|
|
// applySourceMeta deterministically overwrites Source/Author/Published on
|
|
// every "source"-type page with meta — the LLM never controls these fields.
|
|
func applySourceMeta(pages []RawPage, meta sourceMeta) {
|
|
for i := range pages {
|
|
if pages[i].Type != "source" {
|
|
continue
|
|
}
|
|
pages[i].Source = meta.Source
|
|
pages[i].Author = meta.Author
|
|
pages[i].Published = meta.Published
|
|
}
|
|
}
|
|
|
|
func stripFences(s string) string {
|
|
for _, prefix := range []string{"```json\n", "```json\r\n", "```\n", "```\r\n"} {
|
|
if strings.HasPrefix(s, prefix) {
|
|
s = strings.TrimPrefix(s, prefix)
|
|
s = strings.TrimSuffix(strings.TrimSpace(s), "```")
|
|
return strings.TrimSpace(s)
|
|
}
|
|
}
|
|
return s
|
|
}
|