Files
LLM-Server/scripts/ingest-memory.sh
T
haylan eaf11b6d4b fix(scripts): ingest-memory.sh never actually computed embeddings
Two bugs, either one fatal:
- litellm-pgvector's embeddings endpoints take a precomputed vector per
  chunk (they don't call the embedding model themselves) — the script
  posted {content, metadata} with no embedding field, guaranteed 422.
- The batch endpoint expects {"embeddings": [...]}; the script posted a
  bare JSON array as the body.

Now embeds each file's chunks via LiteLLM's /v1/embeddings first (using
LITELLM_PGVECTOR_EMBEDDING_KEY, already minted for exactly this) before
batch-inserting. Also made both source files optional — a missing file is
skipped, not a hard failure, since neither exists in this checkout yet.
2026-09-02 21:32:29 +00:00

86 lines
3.7 KiB
Bash

#!/usr/bin/env bash
# Loads data/memory.md and data/claude-legacy-memory.md into the LiteLLM
# knowledgebase (the "memory-and-notes" vector store, see litellm-config.yaml)
# via litellm-pgvector's batch-embeddings endpoint. Both files are optional —
# a file that doesn't exist yet is skipped, not an error.
#
# ponytail: one chunk per non-empty, non-heading line — both source files are
# already one fact/paragraph per line (no hard-wrapping), so this needs no
# real chunking logic. Re-run after editing either file; there's no dedup, so
# this appends duplicates on a second run against unchanged content — clear
# the store first (DELETE the vector_store_id's rows) if you need a clean
# reload.
set -euo pipefail
cd "$(dirname "$0")/.."
[ -f .env ] && set -a && . ./.env && set +a
: "${LITELLM_PGVECTOR_API_KEY:?Set LITELLM_PGVECTOR_API_KEY in .env first}"
: "${LITELLM_PGVECTOR_EMBEDDING_KEY:?Set LITELLM_PGVECTOR_EMBEDDING_KEY in .env first}"
LITELLM_PGVECTOR_URL="${LITELLM_PGVECTOR_URL:-http://localhost:8000}"
LITELLM_URL="${LITELLM_URL:-http://localhost:${LITELLM_PORT:-4000}}"
VECTOR_STORE_ID="memory-and-notes"
# Must match litellm-config.yaml's vector_store_registry entry — the
# registry just points at a store the backend must already know about.
# Ignores failure if it already exists (no documented idempotency check).
# id is a local addition to litellm-pgvector's create endpoint (see
# vendor/litellm-pgvector/main.py) — without it, create always minted a
# random UUID and this script's writes could never land on VECTOR_STORE_ID.
curl -sf -X POST "${LITELLM_PGVECTOR_URL}/v1/vector_stores" \
-H "Authorization: Bearer ${LITELLM_PGVECTOR_API_KEY}" \
-H "Content-Type: application/json" \
-d "{\"id\": \"${VECTOR_STORE_ID}\", \"name\": \"${VECTOR_STORE_ID}\"}" > /dev/null 2>&1 || true
ingest_file() {
local file="$1"
if [ ! -f "$file" ]; then
echo "Skipping $file (not present)."
return
fi
local section="" contents="[]" metas="[]"
while IFS= read -r line; do
case "$line" in
"#"*) section="${line#\# }"; section="${section#\#\# }"; continue ;;
""|"---") continue ;;
esac
contents=$(jq --arg c "$line" '. += [$c]' <<<"$contents")
metas=$(jq --arg content "$line" --arg source "$file" --arg section "$section" \
'. += [{"content": $content, "metadata": {"source": $source, "section": $section}}]' <<<"$metas")
done < "$file"
local n
n=$(jq 'length' <<<"$contents")
if [ "$n" -eq 0 ]; then
echo "Nothing to ingest from $file (no fact/paragraph lines)."
return
fi
# litellm-pgvector's embeddings endpoints take a precomputed vector per
# chunk — they don't call the embedding model themselves (only query-time
# search does, via its own EMBEDDING__* config). So this has to embed
# client-side first, via the same proxy every other workload uses.
echo "Embedding $n chunks from $file via LiteLLM..."
local embeddings
embeddings=$(curl -sf "${LITELLM_URL}/v1/embeddings" \
-H "Authorization: Bearer ${LITELLM_PGVECTOR_EMBEDDING_KEY}" \
-H "Content-Type: application/json" \
-d "$(jq -n --argjson input "$contents" '{"model": "local-embedding", "input": $input}')" \
| jq '[.data[].embedding]')
local batch
batch=$(jq -n --argjson metas "$metas" --argjson embeds "$embeddings" \
'[range(0; ($metas | length)) as $i | $metas[$i] + {"embedding": $embeds[$i]}]')
echo "Ingesting $n chunks from $file..."
curl -sf -X POST "${LITELLM_PGVECTOR_URL}/v1/vector_stores/${VECTOR_STORE_ID}/embeddings/batch" \
-H "Authorization: Bearer ${LITELLM_PGVECTOR_API_KEY}" \
-H "Content-Type: application/json" \
-d "$(jq -n --argjson embeddings "$batch" '{"embeddings": $embeddings}')" > /dev/null
}
ingest_file data/memory.md
ingest_file data/claude-legacy-memory.md
echo "Done."