Files
Nicolò Boschi 280f098202 feat(retain): inline images and files as first-class content (#4077)
Makes images and files first-class raw content in `retain`. `content` accepts an
ordered list of text/image/file blocks, the extractor reads each attachment in
the position it occupies, and every read surface hands back the attachments
behind what it returns. A plain string behaves exactly as before — text-only
retain is byte-identical, because everything new sits behind an ATTACHMENTS
block that is empty when a chunk carries none.

Blocks are flattened at the API boundary into one canonical body with atomic
placeholders, so `documents.original_text` stays plain text and content_hash
idempotency, `update_mode=append`, chunk-delta re-extraction and
`reprocess_document` keep working untouched. Bytes live in the existing
FileStorage abstraction, content-addressed by sha256.

Schema (one migration, both dialects): `attachments` for the blob,
`document_attachments` for which documents reference it, and
`memory_units.attachment_ids` for which attachments a *fact* came from — a
column rather than a third table, because those ids behave exactly like `tags`.

Provenance is per fact, not per chunk. Extraction runs one call per chunk, and a
chunk holding a screenshot also holds the prose around it, so a chunk-level edge
cited the diagram as evidence for the paragraph that never mentioned it. The
extractor is asked instead, and a fact stated in the prose carries nothing.

Extraction quality was measured against a real image-QA dataset with a raw-VLM
ceiling arm before merging: transcribing structured attachments rather than
summarizing them, and recording how each value is drawn, took the gap between
"the model can read this off the image" and "memory can answer it" from 31.3% to
10.0% on the same 40 charts. The prose-article benchmark went 75% -> 100% over
the same change, so it is not chart-specific tuning.

Also here:

* A vision slot (`HINDSIGHT_API_VLM_*`) so attachment-bearing chunks alone use a
  vision model and text-only chunks stay on a cheaper retain LLM. A vision call
  deliberately does not fail over to the retain chain's text models — that would
  reintroduce the silent omission the 422 gate exists to prevent.
* The extension retain hook can now see each attachment (media type, size, kind,
  filename) and refusing a retain reclaims its bytes, which previously stayed
  fetchable forever.
* A filename lives on the document edge, not the blob: the same PDF can be
  attached under a different name elsewhere, and content-addressing made the
  first name win for both.

Known limitations, documented rather than hidden: store-owned memory backends
get nothing (that retain path is Postgres-free and pre-dates this work), very
dense pages are sampled rather than exhausted, and the Python client's
ContentBlock is a plain dict where TypeScript gets the real union.

Breaking for Go and Rust callers: `content` is now a union, so a bare string no
longer satisfies it. Go gains a `TextContent()` helper; Rust uses
`Content::Variant0(...)`.
2026-09-04 12:48:08 +02:00

138 lines
4.7 KiB
Go

package main
import (
"context"
"fmt"
"net/http"
"os"
"time"
hindsight "github.com/vectorize-io/hindsight/hindsight-clients/go"
)
const kpBankID = "knowledge-pages-demo-bank-go"
func main() {
apiURL := os.Getenv("HINDSIGHT_API_URL")
if apiURL == "" {
apiURL = "http://localhost:8888"
}
cfg := hindsight.NewConfiguration()
cfg.Servers = hindsight.ServerConfigurations{{URL: apiURL}}
client := hindsight.NewAPIClient(cfg)
ctx := context.Background()
// =============================================================================
// Setup (not shown in docs)
// =============================================================================
client.BanksAPI.CreateOrUpdateBank(ctx, kpBankID).
CreateBankRequest(hindsight.CreateBankRequest{
Name: *hindsight.NewNullableString(hindsight.PtrString("Knowledge Pages Demo")),
}).Execute()
for _, content := range []string{
"The API is deployed to Kubernetes with a rolling update",
"Deploys run from the main branch after CI passes",
"A failed deploy is rolled back by redeploying the previous tag",
} {
client.MemoryAPI.RetainMemories(ctx, kpBankID).
RetainRequest(hindsight.RetainRequest{
Items: []hindsight.MemoryItem{{Content: hindsight.TextContent(content)}},
}).Execute()
}
time.Sleep(2 * time.Second)
// [docs:create-folder]
// Create a folder (leave ParentId unset to create it at the root)
folder, _, _ := client.KnowledgeBaseAPI.CreateKnowledgeFolder(ctx, kpBankID).
CreateFolderRequest(hindsight.CreateFolderRequest{Name: "Operations"}).
Execute()
fmt.Printf("Folder ID: %s\n", folder.Id)
// [/docs:create-folder]
// [docs:create-page]
// Create a page — content is generated in the background
page, _, _ := client.KnowledgeBaseAPI.CreateKnowledgePage(ctx, kpBankID).
CreatePageRequest(hindsight.CreatePageRequest{
Name: "Deploying the API",
SourceQuery: "How is the API deployed?",
ParentId: *hindsight.NewNullableString(&folder.Id),
Tags: []string{"ops", "type:runbook"},
}).Execute()
// Poll the operation to know when the first build has finished
fmt.Printf("Page ID: %s, operation: %s\n", page.PageId, page.GetOperationId())
// [/docs:create-page]
// Wait for the page's first build
time.Sleep(20 * time.Second)
// [docs:get-tree]
// Fetch the whole knowledge base as a nested folder/page tree (no page bodies)
tree, _, _ := client.KnowledgeBaseAPI.GetKnowledgeBaseTree(ctx, kpBankID).Execute()
for _, root := range tree.Roots {
fmt.Printf("%s: %s\n", root.Kind, root.Name)
for _, child := range root.Children {
fmt.Printf(" %s: %s (stale: %v)\n", child.Kind, child.Name, child.GetIsStale())
}
}
// [/docs:get-tree]
// [docs:get-page]
// Read a page as a markdown document
document, _, _ := client.KnowledgeBaseAPI.GetKnowledgePage(ctx, kpBankID, page.PageId).Execute()
fmt.Println(document.Type) // "runbook" — from the type:runbook tag
fmt.Println(document.GetBody()) // the synthesized markdown body
fmt.Println(document.Markdown) // YAML frontmatter + body
// [/docs:get-page]
// [docs:search-pages]
// Hybrid search (full-text + vector) over whole pages
results, _, _ := client.KnowledgeBaseAPI.SearchKnowledgeBase(ctx, kpBankID).
Q("how do we deploy").Limit(5).Execute()
for _, hit := range results.Results {
fmt.Printf("%.3f %s: %s\n", hit.Score, hit.Name, hit.Snippet)
}
// [/docs:search-pages]
// [docs:update-node]
// Rename a node, move it, and/or update a page's options.
// Changing SourceQuery rebuilds the page against the new question.
client.KnowledgeBaseAPI.UpdateKnowledgeNode(ctx, kpBankID, page.PageId).
UpdateNodeRequest(hindsight.UpdateNodeRequest{
Name: *hindsight.NewNullableString(hindsight.PtrString("Deploying the API (v2)")),
Tags: []string{"ops", "type:runbook", "reviewed"},
}).Execute()
// [/docs:update-node]
// [docs:export]
// Export the knowledge base as a portable markdown bundle
bundle, _, _ := client.KnowledgeBaseAPI.ExportKnowledgeBase(ctx, kpBankID).Execute()
for _, file := range bundle.Files {
fmt.Println(file.Path) // index.md, <page-id>.md, <page-id>.log.md
}
// [/docs:export]
// [docs:delete-node]
// Delete a folder or page — deleting a folder removes its whole subtree
client.KnowledgeBaseAPI.DeleteKnowledgeNode(ctx, kpBankID, folder.Id).Execute()
// [/docs:delete-node]
// =============================================================================
// Cleanup (not shown in docs)
// =============================================================================
cleanupKnowledgePages(apiURL)
fmt.Println("knowledge-pages.go: All examples passed")
}
func cleanupKnowledgePages(apiURL string) {
req, _ := http.NewRequest("DELETE", fmt.Sprintf("%s/v1/default/banks/%s", apiURL, kpBankID), nil)
http.DefaultClient.Do(req)
}