#!/usr/bin/env bash # Loads data/memory.md and data/claude-legacy-memory.md into the LiteLLM # knowledgebase (the "memory-and-notes" vector store, see litellm-config.yaml) # via litellm-pgvector's batch-embeddings endpoint. Both files are optional — # a file that doesn't exist yet is skipped, not an error. # # ponytail: one chunk per non-empty, non-heading line — both source files are # already one fact/paragraph per line (no hard-wrapping), so this needs no # real chunking logic. Re-run after editing either file; there's no dedup, so # this appends duplicates on a second run against unchanged content — clear # the store first (DELETE the vector_store_id's rows) if you need a clean # reload. set -euo pipefail cd "$(dirname "$0")/.." [ -f .env ] && set -a && . ./.env && set +a : "${LITELLM_PGVECTOR_API_KEY:?Set LITELLM_PGVECTOR_API_KEY in .env first}" : "${LITELLM_PGVECTOR_EMBEDDING_KEY:?Set LITELLM_PGVECTOR_EMBEDDING_KEY in .env first}" LITELLM_PGVECTOR_URL="${LITELLM_PGVECTOR_URL:-http://localhost:8000}" LITELLM_URL="${LITELLM_URL:-http://localhost:${LITELLM_PORT:-4000}}" VECTOR_STORE_ID="memory-and-notes" # Must match litellm-config.yaml's vector_store_registry entry — the # registry just points at a store the backend must already know about. # Ignores failure if it already exists (no documented idempotency check). # id is a local addition to litellm-pgvector's create endpoint (see # vendor/litellm-pgvector/main.py) — without it, create always minted a # random UUID and this script's writes could never land on VECTOR_STORE_ID. curl -sf -X POST "${LITELLM_PGVECTOR_URL}/v1/vector_stores" \ -H "Authorization: Bearer ${LITELLM_PGVECTOR_API_KEY}" \ -H "Content-Type: application/json" \ -d "{\"id\": \"${VECTOR_STORE_ID}\", \"name\": \"${VECTOR_STORE_ID}\"}" > /dev/null 2>&1 || true ingest_file() { local file="$1" if [ ! -f "$file" ]; then echo "Skipping $file (not present)." return fi local section="" contents="[]" metas="[]" while IFS= read -r line; do case "$line" in "#"*) section="${line#\# }"; section="${section#\#\# }"; continue ;; ""|"---") continue ;; esac contents=$(jq --arg c "$line" '. += [$c]' <<<"$contents") metas=$(jq --arg content "$line" --arg source "$file" --arg section "$section" \ '. += [{"content": $content, "metadata": {"source": $source, "section": $section}}]' <<<"$metas") done < "$file" local n n=$(jq 'length' <<<"$contents") if [ "$n" -eq 0 ]; then echo "Nothing to ingest from $file (no fact/paragraph lines)." return fi # litellm-pgvector's embeddings endpoints take a precomputed vector per # chunk — they don't call the embedding model themselves (only query-time # search does, via its own EMBEDDING__* config). So this has to embed # client-side first, via the same proxy every other workload uses. echo "Embedding $n chunks from $file via LiteLLM..." local embeddings embeddings=$(curl -sf "${LITELLM_URL}/v1/embeddings" \ -H "Authorization: Bearer ${LITELLM_PGVECTOR_EMBEDDING_KEY}" \ -H "Content-Type: application/json" \ -d "$(jq -n --argjson input "$contents" '{"model": "local-embedding", "input": $input}')" \ | jq '[.data[].embedding]') local batch batch=$(jq -n --argjson metas "$metas" --argjson embeds "$embeddings" \ '[range(0; ($metas | length)) as $i | $metas[$i] + {"embedding": $embeds[$i]}]') echo "Ingesting $n chunks from $file..." curl -sf -X POST "${LITELLM_PGVECTOR_URL}/v1/vector_stores/${VECTOR_STORE_ID}/embeddings/batch" \ -H "Authorization: Bearer ${LITELLM_PGVECTOR_API_KEY}" \ -H "Content-Type: application/json" \ -d "$(jq -n --argjson embeddings "$batch" '{"embeddings": $embeddings}')" > /dev/null } ingest_file data/memory.md ingest_file data/claude-legacy-memory.md echo "Done."