Files
mathiasandClaude Sonnet 5 6ad275b505
CI / Lint / Test / Vet (push) Failing after 1s
CI / Mirror to GitHub (push) Has been skipped
fix(ingest): thread source/author/published frontmatter through extraction (#85)
pipeline.Run parses source/author/published from the raw content's own
frontmatter and applies them deterministically to source-type RawPages
after LLM extraction — never relies on the LLM to copy them through.
buildFrontmatter now emits all three (source pages only) when present.

Backfill of the 46 already-affected wiki/sources/*.md notes is a
separate concern per the issue, not done here.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Roq1ajWKR5f1hG5Df9wC6A
2026-07-26 23:42:33 +02:00

173 lines
5.1 KiB
Go

// ingestion/internal/pipeline/parse.go
package pipeline
import (
"encoding/json"
"fmt"
"strings"
)
// RawPage is the LLM's output format — minimal structured data with no path or frontmatter.
// The pipeline derives slugs, paths, and frontmatter from these fields.
type RawPage struct {
Title string `json:"title"`
Type string `json:"type"` // "source" | "concept" | "entity"
Subtype string `json:"subtype"` // entity: person|company|tool|model|framework|technology; source: article|pdf|book|video|note|project
Domain string `json:"domain"`
Content string `json:"content"` // Markdown body only — no frontmatter
// Source, Author, Published are deterministic passthrough from the raw
// ingested content's own frontmatter (see parseContentFrontmatter) — never
// set by the LLM. json:"-" keeps them immune to same-named keys the LLM
// might emit. Only meaningful for Type == "source".
Source string `json:"-"`
Author string `json:"-"`
Published string `json:"-"`
}
// ParseRawPages parses LLM output as a JSON array of RawPage objects.
// If the output contains invalid JSON escape sequences (e.g. \. from Markdown),
// it attempts repair before falling back to truncation recovery.
func ParseRawPages(output string) ([]RawPage, []string) {
output = strings.TrimSpace(output)
if output == "" {
return nil, []string{"LLM returned empty output"}
}
output = stripFences(output)
// Fast path: valid JSON.
var pages []RawPage
if err := json.Unmarshal([]byte(output), &pages); err == nil {
return pages, nil
}
// Repair pass: fix invalid escape sequences (e.g. \. \d from Markdown content).
repaired := repairJSON(output)
if err := json.Unmarshal([]byte(repaired), &pages); err == nil {
return pages, []string{"repaired invalid JSON escape sequences in LLM output"}
}
// Truncation recovery: find last `}` that closes a complete object.
idx := strings.LastIndex(repaired, "}")
if idx < 0 {
return nil, []string{"LLM output contained no complete JSON objects"}
}
start := strings.Index(repaired, "[")
if start < 0 {
return nil, []string{"LLM output contained no JSON array opening bracket"}
}
candidate := repaired[start:idx+1] + "]"
if err := json.Unmarshal([]byte(candidate), &pages); err != nil {
return nil, []string{fmt.Sprintf("truncation recovery failed: %v", err)}
}
return pages, []string{fmt.Sprintf("LLM output was truncated; recovered %d page(s)", len(pages))}
}
// repairJSON replaces invalid JSON escape sequences (e.g. \. \d \p) with
// a properly escaped backslash followed by the same character.
// It iterates byte-by-byte to correctly skip already-valid escape sequences
// (including \\) without requiring lookbehind support.
func repairJSON(s string) string {
var b strings.Builder
b.Grow(len(s))
i := 0
for i < len(s) {
if s[i] != '\\' {
b.WriteByte(s[i])
i++
continue
}
// We have a backslash. Peek at the next character.
if i+1 >= len(s) {
// Trailing backslash — emit as-is.
b.WriteByte(s[i])
i++
continue
}
next := s[i+1]
switch next {
case '"', '\\', '/', 'b', 'f', 'n', 'r', 't', 'u':
// Valid JSON escape sequence — emit both characters as-is.
b.WriteByte(s[i])
b.WriteByte(next)
i += 2
default:
// Invalid escape — double the backslash.
b.WriteByte('\\')
b.WriteByte('\\')
b.WriteByte(next)
i += 2
}
}
return b.String()
}
// sourceMeta is source/author/published pulled from the raw ingested
// content's own frontmatter — deterministic passthrough, never LLM output.
type sourceMeta struct {
Source string
Author string
Published string
}
// parseContentFrontmatter extracts source/author/published from a leading
// "---\n...\n---" YAML block in raw ingested content. Only these three flat
// scalar keys are recognised; anything else in the block is ignored. Returns
// a zero-value sourceMeta if content has no frontmatter block.
func parseContentFrontmatter(content string) sourceMeta {
var meta sourceMeta
if !strings.HasPrefix(content, "---\n") && !strings.HasPrefix(content, "---\r\n") {
return meta
}
lines := strings.Split(content, "\n")
for _, line := range lines[1:] {
if strings.TrimSpace(line) == "---" {
break
}
key, val, ok := strings.Cut(line, ":")
if !ok {
continue
}
key = strings.TrimSpace(key)
val = strings.Trim(strings.TrimSpace(val), `"'`)
switch key {
case "source":
meta.Source = val
case "author":
meta.Author = val
case "published":
meta.Published = val
}
}
return meta
}
// applySourceMeta deterministically overwrites Source/Author/Published on
// every "source"-type page with meta — the LLM never controls these fields.
func applySourceMeta(pages []RawPage, meta sourceMeta) {
for i := range pages {
if pages[i].Type != "source" {
continue
}
pages[i].Source = meta.Source
pages[i].Author = meta.Author
pages[i].Published = meta.Published
}
}
func stripFences(s string) string {
for _, prefix := range []string{"```json\n", "```json\r\n", "```\n", "```\r\n"} {
if strings.HasPrefix(s, prefix) {
s = strings.TrimPrefix(s, prefix)
s = strings.TrimSuffix(strings.TrimSpace(s), "```")
return strings.TrimSpace(s)
}
}
return s
}