Compare commits
71
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d67c918695 | ||
|
|
a20dd414dd | ||
|
|
095f462db7 | ||
|
|
470adc6d18 | ||
|
|
11545b6012 | ||
|
|
c369c5d19a | ||
|
|
842f5765a1 | ||
|
|
3eb74c2da4 | ||
|
|
3e482337c7 | ||
|
|
c41665861e | ||
|
|
ae7db69f6d | ||
|
|
befe103871 | ||
|
|
d053e27832 | ||
|
|
f903948a26 | ||
|
|
77defd96e9 | ||
|
|
c2876490df | ||
|
|
eaeaa1f24d | ||
|
|
f6f1a7d889 | ||
|
|
ecb833f40d | ||
|
|
5270aa5a6e | ||
|
|
ad1660b313 | ||
|
|
55af468187 | ||
|
|
2b5fb10dab | ||
|
|
5443c18bfc | ||
|
|
1c21c0c1a6 | ||
|
|
145454533c | ||
|
|
9aaa78b9a6 | ||
|
|
bfa09d1685 | ||
|
|
6f5245ae58 | ||
|
|
fed987f931 | ||
|
|
8cd65b9896 | ||
|
|
b813bd2728 | ||
|
|
86d8ac08b1 | ||
|
|
4b328a9cb3 | ||
|
|
f5b94d4b28 | ||
|
|
58f2de70fb | ||
|
|
0bb5ca4caf | ||
|
|
3ffec65090 | ||
|
|
0aa7c2b3a1 | ||
|
|
ac9a94ade3 | ||
|
|
40b02645f4 | ||
|
|
5fddd9a79c | ||
|
|
4d030707ad | ||
|
|
5fef54d501 | ||
|
|
2a32273226 | ||
|
|
87219b731d | ||
|
|
9f0c031df7 | ||
|
|
8b1a46585d | ||
|
|
172596751f | ||
|
|
b180b3ad97 | ||
|
|
7a2798eb7a | ||
|
|
278344b3b3 | ||
|
|
3f9eb27cd7 | ||
|
|
13c82bab60 | ||
|
|
4f431b4ace | ||
|
|
2993fdd2f9 | ||
|
|
325b5cc141 | ||
|
|
0758827d39 | ||
|
|
ea8163c56d | ||
|
|
ac73948706 | ||
|
|
5569d4adba | ||
|
|
e785b05831 | ||
|
|
58c4d65778 | ||
|
|
c3ef1555bf | ||
|
|
dcaa9f14ab | ||
|
|
41db2960c5 | ||
|
|
f78278ea89 | ||
|
|
117dd6988d | ||
|
|
7c78ae2371 | ||
|
|
6c695eb9f8 | ||
|
|
7eafba661e |
+1
-1
@@ -5,7 +5,7 @@
|
||||
# Supported providers: openai, groq, ollama, gemini, anthropic, lmstudio, vertexai
|
||||
HINDSIGHT_API_LLM_PROVIDER=openai
|
||||
HINDSIGHT_API_LLM_API_KEY=your-api-key-here
|
||||
HINDSIGHT_API_LLM_MODEL=o3-mini
|
||||
HINDSIGHT_API_LLM_MODEL=gpt-4o-mini
|
||||
HINDSIGHT_API_LLM_BASE_URL=https://api.openai.com/v1
|
||||
|
||||
# Example: Anthropic Claude configuration
|
||||
|
||||
@@ -31,6 +31,9 @@ jobs:
|
||||
- run: npm ci --workspace=hindsight-docs
|
||||
- run: uv run generate-llms-full
|
||||
- run: npm run build --workspace=hindsight-docs
|
||||
env:
|
||||
UMAMI_URL: https://analytics.hindsight.vectorize.io
|
||||
UMAMI_WEBSITE_ID: ${{ secrets.UMAMI_WEBSITE_ID }}
|
||||
- uses: actions/upload-pages-artifact@v3
|
||||
with:
|
||||
path: hindsight-docs/build
|
||||
|
||||
@@ -46,6 +46,10 @@ jobs:
|
||||
working-directory: ./hindsight-embed
|
||||
run: uv build --out-dir dist
|
||||
|
||||
- name: Build hindsight-crewai
|
||||
working-directory: ./hindsight-integrations/crewai
|
||||
run: uv build --out-dir dist
|
||||
|
||||
# Publish in order (client and api first, then hindsight-all which depends on them)
|
||||
- name: Publish hindsight-client to PyPI
|
||||
uses: pypa/gh-action-pypi-publish@release/v1
|
||||
@@ -77,6 +81,12 @@ jobs:
|
||||
packages-dir: ./hindsight-embed/dist
|
||||
skip-existing: true
|
||||
|
||||
- name: Publish hindsight-crewai to PyPI
|
||||
uses: pypa/gh-action-pypi-publish@release/v1
|
||||
with:
|
||||
packages-dir: ./hindsight-integrations/crewai/dist
|
||||
skip-existing: true
|
||||
|
||||
# Upload artifacts for GitHub release
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v4
|
||||
@@ -88,6 +98,7 @@ jobs:
|
||||
hindsight/dist/*
|
||||
hindsight-integrations/litellm/dist/*
|
||||
hindsight-embed/dist/*
|
||||
hindsight-integrations/crewai/dist/*
|
||||
retention-days: 1
|
||||
|
||||
release-typescript-client:
|
||||
@@ -188,55 +199,6 @@ jobs:
|
||||
path: hindsight-integrations/openclaw/*.tgz
|
||||
retention-days: 1
|
||||
|
||||
release-nemoclaw-integration:
|
||||
runs-on: ubuntu-latest
|
||||
environment: npm
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Node.js
|
||||
uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '22'
|
||||
registry-url: 'https://registry.npmjs.org'
|
||||
|
||||
- name: Install dependencies
|
||||
working-directory: ./hindsight-integrations/nemoclaw
|
||||
run: npm ci
|
||||
|
||||
- name: Build
|
||||
working-directory: ./hindsight-integrations/nemoclaw
|
||||
run: npm run build
|
||||
|
||||
- name: Publish to npm
|
||||
working-directory: ./hindsight-integrations/nemoclaw
|
||||
run: |
|
||||
set +e
|
||||
OUTPUT=$(npm publish --access public 2>&1)
|
||||
EXIT_CODE=$?
|
||||
echo "$OUTPUT"
|
||||
if [ $EXIT_CODE -ne 0 ]; then
|
||||
if echo "$OUTPUT" | grep -q "cannot publish over"; then
|
||||
echo "Package version already published, skipping..."
|
||||
exit 0
|
||||
fi
|
||||
exit $EXIT_CODE
|
||||
fi
|
||||
env:
|
||||
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
|
||||
|
||||
- name: Pack for GitHub release
|
||||
working-directory: ./hindsight-integrations/nemoclaw
|
||||
run: npm pack
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: nemoclaw-integration
|
||||
path: hindsight-integrations/nemoclaw/*.tgz
|
||||
retention-days: 1
|
||||
|
||||
release-ai-sdk-integration:
|
||||
runs-on: ubuntu-latest
|
||||
environment: npm
|
||||
@@ -286,6 +248,55 @@ jobs:
|
||||
path: hindsight-integrations/ai-sdk/*.tgz
|
||||
retention-days: 1
|
||||
|
||||
release-chat-integration:
|
||||
runs-on: ubuntu-latest
|
||||
environment: npm
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Node.js
|
||||
uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '22'
|
||||
registry-url: 'https://registry.npmjs.org'
|
||||
|
||||
- name: Install dependencies
|
||||
working-directory: ./hindsight-integrations/chat
|
||||
run: npm ci
|
||||
|
||||
- name: Build
|
||||
working-directory: ./hindsight-integrations/chat
|
||||
run: npm run build
|
||||
|
||||
- name: Publish to npm
|
||||
working-directory: ./hindsight-integrations/chat
|
||||
run: |
|
||||
set +e
|
||||
OUTPUT=$(npm publish --access public 2>&1)
|
||||
EXIT_CODE=$?
|
||||
echo "$OUTPUT"
|
||||
if [ $EXIT_CODE -ne 0 ]; then
|
||||
if echo "$OUTPUT" | grep -q "cannot publish over"; then
|
||||
echo "Package version already published, skipping..."
|
||||
exit 0
|
||||
fi
|
||||
exit $EXIT_CODE
|
||||
fi
|
||||
env:
|
||||
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
|
||||
|
||||
- name: Pack for GitHub release
|
||||
working-directory: ./hindsight-integrations/chat
|
||||
run: npm pack
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: chat-integration
|
||||
path: hindsight-integrations/chat/*.tgz
|
||||
retention-days: 1
|
||||
|
||||
release-control-plane:
|
||||
runs-on: ubuntu-latest
|
||||
environment: npm
|
||||
@@ -317,11 +328,14 @@ jobs:
|
||||
- name: Build
|
||||
run: npm run build --workspace=hindsight-control-plane
|
||||
|
||||
- name: Verify standalone build
|
||||
run: test -f hindsight-control-plane/standalone/server.js || (echo 'standalone/server.js missing - build failed' && exit 1)
|
||||
|
||||
- name: Publish to npm
|
||||
working-directory: ./hindsight-control-plane
|
||||
run: |
|
||||
set +e
|
||||
OUTPUT=$(npm publish --access public 2>&1)
|
||||
OUTPUT=$(npm publish --access public --ignore-scripts 2>&1)
|
||||
EXIT_CODE=$?
|
||||
echo "$OUTPUT"
|
||||
if [ $EXIT_CODE -ne 0 ]; then
|
||||
@@ -536,7 +550,7 @@ jobs:
|
||||
|
||||
create-github-release:
|
||||
runs-on: ubuntu-latest
|
||||
needs: [release-python-packages, release-typescript-client, release-openclaw-integration, release-nemoclaw-integration, release-ai-sdk-integration, release-control-plane, release-rust-cli, release-docker-images, release-helm-chart]
|
||||
needs: [release-python-packages, release-typescript-client, release-openclaw-integration, release-ai-sdk-integration, release-chat-integration, release-control-plane, release-rust-cli, release-docker-images, release-helm-chart]
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
@@ -565,18 +579,18 @@ jobs:
|
||||
name: openclaw-integration
|
||||
path: ./artifacts/openclaw-integration
|
||||
|
||||
- name: Download NemoClaw Integration
|
||||
uses: actions/download-artifact@v4
|
||||
with:
|
||||
name: nemoclaw-integration
|
||||
path: ./artifacts/nemoclaw-integration
|
||||
|
||||
- name: Download AI SDK Integration
|
||||
uses: actions/download-artifact@v4
|
||||
with:
|
||||
name: ai-sdk-integration
|
||||
path: ./artifacts/ai-sdk-integration
|
||||
|
||||
- name: Download Chat Integration
|
||||
uses: actions/download-artifact@v4
|
||||
with:
|
||||
name: chat-integration
|
||||
path: ./artifacts/chat-integration
|
||||
|
||||
- name: Download Control Plane
|
||||
uses: actions/download-artifact@v4
|
||||
with:
|
||||
@@ -620,10 +634,10 @@ jobs:
|
||||
cp artifacts/typescript-client/*.tgz release-assets/ || true
|
||||
# OpenClaw Integration
|
||||
cp artifacts/openclaw-integration/*.tgz release-assets/ || true
|
||||
# NemoClaw Integration
|
||||
cp artifacts/nemoclaw-integration/*.tgz release-assets/ || true
|
||||
# AI SDK Integration
|
||||
cp artifacts/ai-sdk-integration/*.tgz release-assets/ || true
|
||||
# Chat Integration
|
||||
cp artifacts/chat-integration/*.tgz release-assets/ || true
|
||||
# Control Plane
|
||||
cp artifacts/control-plane/*.tgz release-assets/ || true
|
||||
# Rust CLI binaries
|
||||
|
||||
+335
-75
@@ -97,6 +97,29 @@ jobs:
|
||||
working-directory: ./hindsight-integrations/ai-sdk
|
||||
run: npm run build
|
||||
|
||||
build-chat-integration:
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Node.js
|
||||
uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '22'
|
||||
|
||||
- name: Install dependencies
|
||||
working-directory: ./hindsight-integrations/chat
|
||||
run: npm ci
|
||||
|
||||
- name: Run tests
|
||||
working-directory: ./hindsight-integrations/chat
|
||||
run: npm test
|
||||
|
||||
- name: Build
|
||||
working-directory: ./hindsight-integrations/chat
|
||||
run: npm run build
|
||||
|
||||
build-control-plane:
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
@@ -171,9 +194,9 @@ jobs:
|
||||
test-rust-cli:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
HINDSIGHT_API_LLM_PROVIDER: groq
|
||||
HINDSIGHT_API_LLM_API_KEY: ${{ secrets.GROQ_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL: openai/gpt-oss-20b
|
||||
HINDSIGHT_API_LLM_PROVIDER: vertexai
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY: /tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_MODEL: google/gemini-2.5-flash-lite
|
||||
HINDSIGHT_API_URL: http://localhost:8888
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
UV_INDEX: pytorch=https://download.pytorch.org/whl/cpu
|
||||
@@ -181,6 +204,12 @@ jobs:
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Setup GCP credentials
|
||||
run: |
|
||||
printf '%s' '${{ secrets.GCP_VERTEXAI_CREDENTIALS }}' > /tmp/gcp-credentials.json
|
||||
PROJECT_ID=$(jq -r '.project_id' /tmp/gcp-credentials.json)
|
||||
echo "HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$PROJECT_ID" >> $GITHUB_ENV
|
||||
|
||||
- name: Install Rust
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
@@ -227,25 +256,46 @@ jobs:
|
||||
working-directory: ./hindsight-api
|
||||
run: uv sync --frozen --no-install-project --index-strategy unsafe-best-match
|
||||
|
||||
- name: Cache HuggingFace models
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.cache/huggingface
|
||||
key: ${{ runner.os }}-huggingface-${{ hashFiles('hindsight-api/pyproject.toml') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-huggingface-
|
||||
|
||||
- name: Pre-download models
|
||||
working-directory: ./hindsight-api
|
||||
run: |
|
||||
uv run python -c "
|
||||
from sentence_transformers import SentenceTransformer, CrossEncoder
|
||||
print('Downloading embedding model...')
|
||||
SentenceTransformer('BAAI/bge-small-en-v1.5')
|
||||
print('Downloading cross-encoder model...')
|
||||
CrossEncoder('cross-encoder/ms-marco-MiniLM-L-6-v2')
|
||||
print('Models downloaded successfully')
|
||||
"
|
||||
|
||||
- name: Create .env file
|
||||
run: |
|
||||
cat > .env << EOF
|
||||
HINDSIGHT_API_LLM_PROVIDER=${{ env.HINDSIGHT_API_LLM_PROVIDER }}
|
||||
HINDSIGHT_API_LLM_API_KEY=${{ env.HINDSIGHT_API_LLM_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL=${{ env.HINDSIGHT_API_LLM_MODEL }}
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY=/tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID
|
||||
EOF
|
||||
|
||||
- name: Start API server
|
||||
run: |
|
||||
./scripts/dev/start-api.sh > /tmp/api-server.log 2>&1 &
|
||||
echo "Waiting for API server to be ready..."
|
||||
for i in {1..60}; do
|
||||
for i in {1..120}; do
|
||||
if curl -sf http://localhost:8888/health > /dev/null 2>&1; then
|
||||
echo "API server is ready after ${i}s"
|
||||
break
|
||||
fi
|
||||
if [ $i -eq 60 ]; then
|
||||
echo "API server failed to start after 60s"
|
||||
if [ $i -eq 120 ]; then
|
||||
echo "API server failed to start after 120s"
|
||||
cat /tmp/api-server.log
|
||||
exit 1
|
||||
fi
|
||||
@@ -340,12 +390,21 @@ jobs:
|
||||
|
||||
# Only test slim variants to save disk space (they're much smaller)
|
||||
# Slim variants require external embedding providers
|
||||
- name: Setup GCP credentials for smoke test
|
||||
if: matrix.variant == 'slim'
|
||||
run: |
|
||||
printf '%s' '${{ secrets.GCP_VERTEXAI_CREDENTIALS }}' > /tmp/gcp-credentials.json
|
||||
PROJECT_ID=$(jq -r '.project_id' /tmp/gcp-credentials.json)
|
||||
echo "HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$PROJECT_ID" >> $GITHUB_ENV
|
||||
|
||||
- name: Smoke test - verify container starts
|
||||
if: matrix.variant == 'slim'
|
||||
env:
|
||||
GROQ_API_KEY: ${{ secrets.GROQ_API_KEY }}
|
||||
HINDSIGHT_API_EMBEDDINGS_PROVIDER: openai
|
||||
HINDSIGHT_API_EMBEDDINGS_OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
HINDSIGHT_API_LLM_PROVIDER: vertexai
|
||||
HINDSIGHT_API_LLM_MODEL: google/gemini-2.5-flash-lite
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY: /tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID: ${{ env.HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID }}
|
||||
HINDSIGHT_API_EMBEDDINGS_PROVIDER: cohere
|
||||
HINDSIGHT_API_RERANKER_PROVIDER: cohere
|
||||
HINDSIGHT_API_COHERE_API_KEY: ${{ secrets.COHERE_API_KEY }}
|
||||
run: ./docker/test-image.sh "hindsight-${{ matrix.name }}:test" "${{ matrix.target }}"
|
||||
@@ -353,14 +412,13 @@ jobs:
|
||||
test-api:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
HINDSIGHT_API_LLM_PROVIDER: groq
|
||||
HINDSIGHT_API_LLM_API_KEY: ${{ secrets.GROQ_API_KEY }}
|
||||
GROQ_API_KEY: ${{ secrets.GROQ_API_KEY }}
|
||||
HINDSIGHT_API_LLM_PROVIDER: vertexai
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY: /tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_MODEL: google/gemini-2.5-flash-lite
|
||||
GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
COHERE_API_KEY: ${{ secrets.COHERE_API_KEY }}
|
||||
HINDSIGHT_API_EMBEDDINGS_OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL: openai/gpt-oss-20b
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
# Prefer CPU-only PyTorch in CI (but keep PyPI for everything else)
|
||||
UV_INDEX: pytorch=https://download.pytorch.org/whl/cpu
|
||||
@@ -368,6 +426,12 @@ jobs:
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Setup GCP credentials
|
||||
run: |
|
||||
printf '%s' '${{ secrets.GCP_VERTEXAI_CREDENTIALS }}' > /tmp/gcp-credentials.json
|
||||
PROJECT_ID=$(jq -r '.project_id' /tmp/gcp-credentials.json)
|
||||
echo "HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$PROJECT_ID" >> $GITHUB_ENV
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v5
|
||||
with:
|
||||
@@ -414,9 +478,9 @@ jobs:
|
||||
test-python-client:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
HINDSIGHT_API_LLM_PROVIDER: groq
|
||||
HINDSIGHT_API_LLM_API_KEY: ${{ secrets.GROQ_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL: openai/gpt-oss-20b
|
||||
HINDSIGHT_API_LLM_PROVIDER: vertexai
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY: /tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_MODEL: google/gemini-2.5-flash-lite
|
||||
HINDSIGHT_API_URL: http://localhost:8888
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
# Prefer CPU-only PyTorch in CI (but keep PyPI for everything else)
|
||||
@@ -425,6 +489,12 @@ jobs:
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Setup GCP credentials
|
||||
run: |
|
||||
printf '%s' '${{ secrets.GCP_VERTEXAI_CREDENTIALS }}' > /tmp/gcp-credentials.json
|
||||
PROJECT_ID=$(jq -r '.project_id' /tmp/gcp-credentials.json)
|
||||
echo "HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$PROJECT_ID" >> $GITHUB_ENV
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v5
|
||||
with:
|
||||
@@ -452,25 +522,46 @@ jobs:
|
||||
working-directory: ./hindsight-api
|
||||
run: uv sync --frozen --no-install-project --index-strategy unsafe-best-match
|
||||
|
||||
- name: Cache HuggingFace models
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.cache/huggingface
|
||||
key: ${{ runner.os }}-huggingface-${{ hashFiles('hindsight-api/pyproject.toml') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-huggingface-
|
||||
|
||||
- name: Pre-download models
|
||||
working-directory: ./hindsight-api
|
||||
run: |
|
||||
uv run python -c "
|
||||
from sentence_transformers import SentenceTransformer, CrossEncoder
|
||||
print('Downloading embedding model...')
|
||||
SentenceTransformer('BAAI/bge-small-en-v1.5')
|
||||
print('Downloading cross-encoder model...')
|
||||
CrossEncoder('cross-encoder/ms-marco-MiniLM-L-6-v2')
|
||||
print('Models downloaded successfully')
|
||||
"
|
||||
|
||||
- name: Create .env file
|
||||
run: |
|
||||
cat > .env << EOF
|
||||
HINDSIGHT_API_LLM_PROVIDER=${{ env.HINDSIGHT_API_LLM_PROVIDER }}
|
||||
HINDSIGHT_API_LLM_API_KEY=${{ env.HINDSIGHT_API_LLM_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL=${{ env.HINDSIGHT_API_LLM_MODEL }}
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY=/tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID
|
||||
EOF
|
||||
|
||||
- name: Start API server
|
||||
run: |
|
||||
./scripts/dev/start-api.sh > /tmp/api-server.log 2>&1 &
|
||||
echo "Waiting for API server to be ready..."
|
||||
for i in {1..60}; do
|
||||
for i in {1..120}; do
|
||||
if curl -sf http://localhost:8888/health > /dev/null 2>&1; then
|
||||
echo "API server is ready after ${i}s"
|
||||
break
|
||||
fi
|
||||
if [ $i -eq 60 ]; then
|
||||
echo "API server failed to start after 60s"
|
||||
if [ $i -eq 120 ]; then
|
||||
echo "API server failed to start after 120s"
|
||||
cat /tmp/api-server.log
|
||||
exit 1
|
||||
fi
|
||||
@@ -490,9 +581,9 @@ jobs:
|
||||
test-typescript-client:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
HINDSIGHT_API_LLM_PROVIDER: groq
|
||||
HINDSIGHT_API_LLM_API_KEY: ${{ secrets.GROQ_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL: openai/gpt-oss-20b
|
||||
HINDSIGHT_API_LLM_PROVIDER: vertexai
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY: /tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_MODEL: google/gemini-2.5-flash-lite
|
||||
HINDSIGHT_API_URL: http://localhost:8888
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
# Prefer CPU-only PyTorch in CI (but keep PyPI for everything else)
|
||||
@@ -501,6 +592,12 @@ jobs:
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Setup GCP credentials
|
||||
run: |
|
||||
printf '%s' '${{ secrets.GCP_VERTEXAI_CREDENTIALS }}' > /tmp/gcp-credentials.json
|
||||
PROJECT_ID=$(jq -r '.project_id' /tmp/gcp-credentials.json)
|
||||
echo "HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$PROJECT_ID" >> $GITHUB_ENV
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v5
|
||||
with:
|
||||
@@ -533,25 +630,46 @@ jobs:
|
||||
working-directory: ./hindsight-clients/typescript
|
||||
run: npm run build
|
||||
|
||||
- name: Cache HuggingFace models
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.cache/huggingface
|
||||
key: ${{ runner.os }}-huggingface-${{ hashFiles('hindsight-api/pyproject.toml') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-huggingface-
|
||||
|
||||
- name: Pre-download models
|
||||
working-directory: ./hindsight-api
|
||||
run: |
|
||||
uv run python -c "
|
||||
from sentence_transformers import SentenceTransformer, CrossEncoder
|
||||
print('Downloading embedding model...')
|
||||
SentenceTransformer('BAAI/bge-small-en-v1.5')
|
||||
print('Downloading cross-encoder model...')
|
||||
CrossEncoder('cross-encoder/ms-marco-MiniLM-L-6-v2')
|
||||
print('Models downloaded successfully')
|
||||
"
|
||||
|
||||
- name: Create .env file
|
||||
run: |
|
||||
cat > .env << EOF
|
||||
HINDSIGHT_API_LLM_PROVIDER=${{ env.HINDSIGHT_API_LLM_PROVIDER }}
|
||||
HINDSIGHT_API_LLM_API_KEY=${{ env.HINDSIGHT_API_LLM_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL=${{ env.HINDSIGHT_API_LLM_MODEL }}
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY=/tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID
|
||||
EOF
|
||||
|
||||
- name: Start API server
|
||||
run: |
|
||||
./scripts/dev/start-api.sh > /tmp/api-server.log 2>&1 &
|
||||
echo "Waiting for API server to be ready..."
|
||||
for i in {1..60}; do
|
||||
for i in {1..120}; do
|
||||
if curl -sf http://localhost:8888/health > /dev/null 2>&1; then
|
||||
echo "API server is ready after ${i}s"
|
||||
break
|
||||
fi
|
||||
if [ $i -eq 60 ]; then
|
||||
echo "API server failed to start after 60s"
|
||||
if [ $i -eq 120 ]; then
|
||||
echo "API server failed to start after 120s"
|
||||
cat /tmp/api-server.log
|
||||
exit 1
|
||||
fi
|
||||
@@ -571,9 +689,9 @@ jobs:
|
||||
test-rust-client:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
HINDSIGHT_API_LLM_PROVIDER: groq
|
||||
HINDSIGHT_API_LLM_API_KEY: ${{ secrets.GROQ_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL: openai/gpt-oss-20b
|
||||
HINDSIGHT_API_LLM_PROVIDER: vertexai
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY: /tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_MODEL: google/gemini-2.5-flash-lite
|
||||
HINDSIGHT_API_URL: http://localhost:8888
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
# Prefer CPU-only PyTorch in CI (but keep PyPI for everything else)
|
||||
@@ -582,6 +700,12 @@ jobs:
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Setup GCP credentials
|
||||
run: |
|
||||
printf '%s' '${{ secrets.GCP_VERTEXAI_CREDENTIALS }}' > /tmp/gcp-credentials.json
|
||||
PROJECT_ID=$(jq -r '.project_id' /tmp/gcp-credentials.json)
|
||||
echo "HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$PROJECT_ID" >> $GITHUB_ENV
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v5
|
||||
with:
|
||||
@@ -613,25 +737,46 @@ jobs:
|
||||
working-directory: ./hindsight-api
|
||||
run: uv sync --frozen --no-install-project --index-strategy unsafe-best-match
|
||||
|
||||
- name: Cache HuggingFace models
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.cache/huggingface
|
||||
key: ${{ runner.os }}-huggingface-${{ hashFiles('hindsight-api/pyproject.toml') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-huggingface-
|
||||
|
||||
- name: Pre-download models
|
||||
working-directory: ./hindsight-api
|
||||
run: |
|
||||
uv run python -c "
|
||||
from sentence_transformers import SentenceTransformer, CrossEncoder
|
||||
print('Downloading embedding model...')
|
||||
SentenceTransformer('BAAI/bge-small-en-v1.5')
|
||||
print('Downloading cross-encoder model...')
|
||||
CrossEncoder('cross-encoder/ms-marco-MiniLM-L-6-v2')
|
||||
print('Models downloaded successfully')
|
||||
"
|
||||
|
||||
- name: Create .env file
|
||||
run: |
|
||||
cat > .env << EOF
|
||||
HINDSIGHT_API_LLM_PROVIDER=${{ env.HINDSIGHT_API_LLM_PROVIDER }}
|
||||
HINDSIGHT_API_LLM_API_KEY=${{ env.HINDSIGHT_API_LLM_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL=${{ env.HINDSIGHT_API_LLM_MODEL }}
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY=/tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID
|
||||
EOF
|
||||
|
||||
- name: Start API server
|
||||
run: |
|
||||
./scripts/dev/start-api.sh > /tmp/api-server.log 2>&1 &
|
||||
echo "Waiting for API server to be ready..."
|
||||
for i in {1..60}; do
|
||||
for i in {1..120}; do
|
||||
if curl -sf http://localhost:8888/health > /dev/null 2>&1; then
|
||||
echo "API server is ready after ${i}s"
|
||||
break
|
||||
fi
|
||||
if [ $i -eq 60 ]; then
|
||||
echo "API server failed to start after 60s"
|
||||
if [ $i -eq 120 ]; then
|
||||
echo "API server failed to start after 120s"
|
||||
cat /tmp/api-server.log
|
||||
exit 1
|
||||
fi
|
||||
@@ -651,9 +796,9 @@ jobs:
|
||||
test-go-client:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
HINDSIGHT_API_LLM_PROVIDER: groq
|
||||
HINDSIGHT_API_LLM_API_KEY: ${{ secrets.GROQ_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL: openai/gpt-oss-20b
|
||||
HINDSIGHT_API_LLM_PROVIDER: vertexai
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY: /tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_MODEL: google/gemini-2.5-flash-lite
|
||||
HINDSIGHT_API_URL: http://localhost:8888
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
# Prefer CPU-only PyTorch in CI (but keep PyPI for everything else)
|
||||
@@ -662,6 +807,12 @@ jobs:
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Setup GCP credentials
|
||||
run: |
|
||||
printf '%s' '${{ secrets.GCP_VERTEXAI_CREDENTIALS }}' > /tmp/gcp-credentials.json
|
||||
PROJECT_ID=$(jq -r '.project_id' /tmp/gcp-credentials.json)
|
||||
echo "HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$PROJECT_ID" >> $GITHUB_ENV
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v5
|
||||
with:
|
||||
@@ -687,25 +838,46 @@ jobs:
|
||||
working-directory: ./hindsight-api
|
||||
run: uv sync --frozen --no-install-project --index-strategy unsafe-best-match
|
||||
|
||||
- name: Cache HuggingFace models
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.cache/huggingface
|
||||
key: ${{ runner.os }}-huggingface-${{ hashFiles('hindsight-api/pyproject.toml') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-huggingface-
|
||||
|
||||
- name: Pre-download models
|
||||
working-directory: ./hindsight-api
|
||||
run: |
|
||||
uv run python -c "
|
||||
from sentence_transformers import SentenceTransformer, CrossEncoder
|
||||
print('Downloading embedding model...')
|
||||
SentenceTransformer('BAAI/bge-small-en-v1.5')
|
||||
print('Downloading cross-encoder model...')
|
||||
CrossEncoder('cross-encoder/ms-marco-MiniLM-L-6-v2')
|
||||
print('Models downloaded successfully')
|
||||
"
|
||||
|
||||
- name: Create .env file
|
||||
run: |
|
||||
cat > .env << EOF
|
||||
HINDSIGHT_API_LLM_PROVIDER=${{ env.HINDSIGHT_API_LLM_PROVIDER }}
|
||||
HINDSIGHT_API_LLM_API_KEY=${{ env.HINDSIGHT_API_LLM_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL=${{ env.HINDSIGHT_API_LLM_MODEL }}
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY=/tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID
|
||||
EOF
|
||||
|
||||
- name: Start API server
|
||||
run: |
|
||||
./scripts/dev/start-api.sh > /tmp/api-server.log 2>&1 &
|
||||
echo "Waiting for API server to be ready..."
|
||||
for i in {1..60}; do
|
||||
for i in {1..120}; do
|
||||
if curl -sf http://localhost:8888/health > /dev/null 2>&1; then
|
||||
echo "API server is ready after ${i}s"
|
||||
break
|
||||
fi
|
||||
if [ $i -eq 60 ]; then
|
||||
echo "API server failed to start after 60s"
|
||||
if [ $i -eq 120 ]; then
|
||||
echo "API server failed to start after 120s"
|
||||
cat /tmp/api-server.log
|
||||
exit 1
|
||||
fi
|
||||
@@ -729,9 +901,9 @@ jobs:
|
||||
test-openclaw-integration:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
HINDSIGHT_API_LLM_PROVIDER: groq
|
||||
HINDSIGHT_API_LLM_API_KEY: ${{ secrets.GROQ_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL: openai/gpt-oss-20b
|
||||
HINDSIGHT_API_LLM_PROVIDER: vertexai
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY: /tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_MODEL: google/gemini-2.5-flash-lite
|
||||
HINDSIGHT_API_URL: http://localhost:8888
|
||||
HINDSIGHT_EMBED_PACKAGE_PATH: ${{ github.workspace }}/hindsight-embed
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
@@ -740,6 +912,12 @@ jobs:
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Setup GCP credentials
|
||||
run: |
|
||||
printf '%s' '${{ secrets.GCP_VERTEXAI_CREDENTIALS }}' > /tmp/gcp-credentials.json
|
||||
PROJECT_ID=$(jq -r '.project_id' /tmp/gcp-credentials.json)
|
||||
echo "HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$PROJECT_ID" >> $GITHUB_ENV
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v5
|
||||
with:
|
||||
@@ -796,21 +974,22 @@ jobs:
|
||||
run: |
|
||||
cat > .env << EOF
|
||||
HINDSIGHT_API_LLM_PROVIDER=${{ env.HINDSIGHT_API_LLM_PROVIDER }}
|
||||
HINDSIGHT_API_LLM_API_KEY=${{ env.HINDSIGHT_API_LLM_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL=${{ env.HINDSIGHT_API_LLM_MODEL }}
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY=/tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID
|
||||
EOF
|
||||
|
||||
- name: Start API server
|
||||
run: |
|
||||
./scripts/dev/start-api.sh > /tmp/api-server.log 2>&1 &
|
||||
echo "Waiting for API server to be ready..."
|
||||
for i in {1..60}; do
|
||||
for i in {1..120}; do
|
||||
if curl -sf http://localhost:8888/health > /dev/null 2>&1; then
|
||||
echo "API server is ready after ${i}s"
|
||||
break
|
||||
fi
|
||||
if [ $i -eq 60 ]; then
|
||||
echo "API server failed to start after 60s"
|
||||
if [ $i -eq 120 ]; then
|
||||
echo "API server failed to start after 120s"
|
||||
cat /tmp/api-server.log
|
||||
exit 1
|
||||
fi
|
||||
@@ -830,9 +1009,9 @@ jobs:
|
||||
test-integration:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
HINDSIGHT_API_LLM_PROVIDER: groq
|
||||
HINDSIGHT_API_LLM_API_KEY: ${{ secrets.GROQ_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL: openai/gpt-oss-20b
|
||||
HINDSIGHT_API_LLM_PROVIDER: vertexai
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY: /tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_MODEL: google/gemini-2.5-flash-lite
|
||||
HINDSIGHT_API_URL: http://localhost:8888
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
UV_INDEX: pytorch=https://download.pytorch.org/whl/cpu
|
||||
@@ -840,6 +1019,12 @@ jobs:
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Setup GCP credentials
|
||||
run: |
|
||||
printf '%s' '${{ secrets.GCP_VERTEXAI_CREDENTIALS }}' > /tmp/gcp-credentials.json
|
||||
PROJECT_ID=$(jq -r '.project_id' /tmp/gcp-credentials.json)
|
||||
echo "HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$PROJECT_ID" >> $GITHUB_ENV
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v5
|
||||
with:
|
||||
@@ -887,21 +1072,22 @@ jobs:
|
||||
run: |
|
||||
cat > .env << EOF
|
||||
HINDSIGHT_API_LLM_PROVIDER=${{ env.HINDSIGHT_API_LLM_PROVIDER }}
|
||||
HINDSIGHT_API_LLM_API_KEY=${{ env.HINDSIGHT_API_LLM_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL=${{ env.HINDSIGHT_API_LLM_MODEL }}
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY=/tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID
|
||||
EOF
|
||||
|
||||
- name: Start API server
|
||||
run: |
|
||||
./scripts/dev/start-api.sh > /tmp/api-server.log 2>&1 &
|
||||
echo "Waiting for API server to be ready..."
|
||||
for i in {1..60}; do
|
||||
for i in {1..120}; do
|
||||
if curl -sf http://localhost:8888/health > /dev/null 2>&1; then
|
||||
echo "API server is ready after ${i}s"
|
||||
break
|
||||
fi
|
||||
if [ $i -eq 60 ]; then
|
||||
echo "API server failed to start after 60s"
|
||||
if [ $i -eq 120 ]; then
|
||||
echo "API server failed to start after 120s"
|
||||
cat /tmp/api-server.log
|
||||
exit 1
|
||||
fi
|
||||
@@ -918,6 +1104,35 @@ jobs:
|
||||
echo "=== API Server Logs ==="
|
||||
cat /tmp/api-server.log || echo "No API server log found"
|
||||
|
||||
test-crewai-integration:
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v5
|
||||
with:
|
||||
enable-cache: true
|
||||
prune-cache: false
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version-file: ".python-version"
|
||||
|
||||
- name: Build crewai integration
|
||||
working-directory: ./hindsight-integrations/crewai
|
||||
run: uv build
|
||||
|
||||
- name: Install dependencies
|
||||
working-directory: ./hindsight-integrations/crewai
|
||||
run: uv sync --frozen
|
||||
|
||||
- name: Run tests
|
||||
working-directory: ./hindsight-integrations/crewai
|
||||
run: uv run pytest tests -v
|
||||
|
||||
test-litellm-integration:
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
@@ -950,15 +1165,21 @@ jobs:
|
||||
test-embed:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
HINDSIGHT_API_LLM_PROVIDER: groq
|
||||
HINDSIGHT_API_LLM_API_KEY: ${{ secrets.GROQ_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL: openai/gpt-oss-20b
|
||||
HINDSIGHT_API_LLM_PROVIDER: vertexai
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY: /tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_MODEL: google/gemini-2.5-flash-lite
|
||||
# Prefer CPU-only PyTorch in CI
|
||||
UV_INDEX: pytorch=https://download.pytorch.org/whl/cpu
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Setup GCP credentials
|
||||
run: |
|
||||
printf '%s' '${{ secrets.GCP_VERTEXAI_CREDENTIALS }}' > /tmp/gcp-credentials.json
|
||||
PROJECT_ID=$(jq -r '.project_id' /tmp/gcp-credentials.json)
|
||||
echo "HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$PROJECT_ID" >> $GITHUB_ENV
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v5
|
||||
with:
|
||||
@@ -994,19 +1215,25 @@ jobs:
|
||||
test-hindsight-all:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
HINDSIGHT_API_LLM_PROVIDER: groq
|
||||
HINDSIGHT_API_LLM_API_KEY: ${{ secrets.GROQ_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL: openai/gpt-oss-20b
|
||||
HINDSIGHT_API_LLM_PROVIDER: vertexai
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY: /tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_MODEL: google/gemini-2.5-flash-lite
|
||||
# For test_server_integration.py compatibility
|
||||
HINDSIGHT_LLM_PROVIDER: groq
|
||||
HINDSIGHT_LLM_API_KEY: ${{ secrets.GROQ_API_KEY }}
|
||||
HINDSIGHT_LLM_MODEL: openai/gpt-oss-20b
|
||||
HINDSIGHT_LLM_PROVIDER: vertexai
|
||||
HINDSIGHT_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY: /tmp/gcp-credentials.json
|
||||
HINDSIGHT_LLM_MODEL: google/gemini-2.5-flash-lite
|
||||
# Prefer CPU-only PyTorch in CI
|
||||
UV_INDEX: pytorch=https://download.pytorch.org/whl/cpu
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Setup GCP credentials
|
||||
run: |
|
||||
printf '%s' '${{ secrets.GCP_VERTEXAI_CREDENTIALS }}' > /tmp/gcp-credentials.json
|
||||
PROJECT_ID=$(jq -r '.project_id' /tmp/gcp-credentials.json)
|
||||
echo "HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$PROJECT_ID" >> $GITHUB_ENV
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v5
|
||||
with:
|
||||
@@ -1043,9 +1270,9 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
needs: test-rust-cli
|
||||
env:
|
||||
HINDSIGHT_API_LLM_PROVIDER: groq
|
||||
HINDSIGHT_API_LLM_API_KEY: ${{ secrets.GROQ_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL: openai/gpt-oss-20b
|
||||
HINDSIGHT_API_LLM_PROVIDER: vertexai
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY: /tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_MODEL: google/gemini-2.5-flash-lite
|
||||
HINDSIGHT_API_URL: http://localhost:8888
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
UV_INDEX: pytorch=https://download.pytorch.org/whl/cpu
|
||||
@@ -1053,6 +1280,12 @@ jobs:
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Setup GCP credentials
|
||||
run: |
|
||||
printf '%s' '${{ secrets.GCP_VERTEXAI_CREDENTIALS }}' > /tmp/gcp-credentials.json
|
||||
PROJECT_ID=$(jq -r '.project_id' /tmp/gcp-credentials.json)
|
||||
echo "HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$PROJECT_ID" >> $GITHUB_ENV
|
||||
|
||||
- name: Download CLI artifact
|
||||
uses: actions/download-artifact@v4
|
||||
with:
|
||||
@@ -1095,25 +1328,46 @@ jobs:
|
||||
npm ci --workspace=hindsight-clients/typescript
|
||||
npm run build --workspace=hindsight-clients/typescript
|
||||
|
||||
- name: Cache HuggingFace models
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.cache/huggingface
|
||||
key: ${{ runner.os }}-huggingface-${{ hashFiles('hindsight-api/pyproject.toml') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-huggingface-
|
||||
|
||||
- name: Pre-download models
|
||||
working-directory: ./hindsight-api
|
||||
run: |
|
||||
uv run python -c "
|
||||
from sentence_transformers import SentenceTransformer, CrossEncoder
|
||||
print('Downloading embedding model...')
|
||||
SentenceTransformer('BAAI/bge-small-en-v1.5')
|
||||
print('Downloading reranker model...')
|
||||
CrossEncoder('cross-encoder/ms-marco-MiniLM-L-6-v2')
|
||||
print('Models downloaded successfully')
|
||||
"
|
||||
|
||||
- name: Create .env file
|
||||
run: |
|
||||
cat > .env << EOF
|
||||
HINDSIGHT_API_LLM_PROVIDER=${{ env.HINDSIGHT_API_LLM_PROVIDER }}
|
||||
HINDSIGHT_API_LLM_API_KEY=${{ env.HINDSIGHT_API_LLM_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL=${{ env.HINDSIGHT_API_LLM_MODEL }}
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY=/tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID
|
||||
EOF
|
||||
|
||||
- name: Start API server
|
||||
run: |
|
||||
./scripts/dev/start-api.sh > /tmp/api-server.log 2>&1 &
|
||||
echo "Waiting for API server to be ready..."
|
||||
for i in {1..60}; do
|
||||
for i in {1..120}; do
|
||||
if curl -sf http://localhost:8888/health > /dev/null 2>&1; then
|
||||
echo "API server is ready after ${i}s"
|
||||
break
|
||||
fi
|
||||
if [ $i -eq 60 ]; then
|
||||
echo "API server failed to start after 60s"
|
||||
if [ $i -eq 120 ]; then
|
||||
echo "API server failed to start after 120s"
|
||||
cat /tmp/api-server.log
|
||||
exit 1
|
||||
fi
|
||||
@@ -1135,9 +1389,9 @@ jobs:
|
||||
test-upgrade:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
HINDSIGHT_API_LLM_PROVIDER: groq
|
||||
HINDSIGHT_API_LLM_API_KEY: ${{ secrets.GROQ_API_KEY }}
|
||||
HINDSIGHT_API_LLM_MODEL: openai/gpt-oss-20b
|
||||
HINDSIGHT_API_LLM_PROVIDER: vertexai
|
||||
HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY: /tmp/gcp-credentials.json
|
||||
HINDSIGHT_API_LLM_MODEL: google/gemini-2.5-flash-lite
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
UV_INDEX: pytorch=https://download.pytorch.org/whl/cpu
|
||||
|
||||
@@ -1146,6 +1400,12 @@ jobs:
|
||||
with:
|
||||
fetch-depth: 0 # Full history needed for git clone of tags
|
||||
|
||||
- name: Setup GCP credentials
|
||||
run: |
|
||||
printf '%s' '${{ secrets.GCP_VERTEXAI_CREDENTIALS }}' > /tmp/gcp-credentials.json
|
||||
PROJECT_ID=$(jq -r '.project_id' /tmp/gcp-credentials.json)
|
||||
echo "HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=$PROJECT_ID" >> $GITHUB_ENV
|
||||
|
||||
- name: Fetch tags
|
||||
run: git fetch --tags
|
||||
|
||||
|
||||
@@ -317,10 +317,10 @@ npm install
|
||||
Required env vars:
|
||||
- `HINDSIGHT_API_LLM_PROVIDER`: openai, anthropic, gemini, groq, ollama, lmstudio
|
||||
- `HINDSIGHT_API_LLM_API_KEY`: Your API key
|
||||
- `HINDSIGHT_API_LLM_MODEL`: Model name (e.g., o3-mini, claude-sonnet-4-20250514)
|
||||
- `HINDSIGHT_API_LLM_MODEL`: Model name (e.g., gpt-4o-mini, claude-sonnet-4-20250514)
|
||||
|
||||
Optional (uses local models by default):
|
||||
- `HINDSIGHT_API_EMBEDDINGS_PROVIDER`: local (default) or tei
|
||||
- `HINDSIGHT_API_RERANKER_PROVIDER`: local (default) or tei
|
||||
- `HINDSIGHT_API_DATABASE_URL`: External PostgreSQL (uses embedded pg0 by default)
|
||||
- `HINDSIGHT_API_ENABLE_BANK_CONFIG_API`: Enable per-bank config API (default: false, disabled for security)
|
||||
- `HINDSIGHT_API_ENABLE_BANK_CONFIG_API`: Enable per-bank config API (default: true)
|
||||
|
||||
@@ -36,7 +36,7 @@ Hindsight is being used in production at Fortune 500 enterprises and by a growin
|
||||
|
||||
## Adding Hindsight to Your AI Agents
|
||||
|
||||
The easiest way use Hindsight with an existing agent is with the LLM Wrapper. You can add memory to your agent with 2 lines of code. That will swap your current LLM client out with the Hindsight wrapper. After that, memories will be stored and retrieved automatically as you make LLM calls.
|
||||
The easiest way to use Hindsight with an existing agent is with the LLM Wrapper. You can add memory to your agent with 2 lines of code. That will swap your current LLM client out with the Hindsight wrapper. After that, memories will be stored and retrieved automatically as you make LLM calls.
|
||||
|
||||
If you need more control over how and when your agent stores and recalls memories, there's also a simple API you can integrate with using the SDKs or directly via HTTP.
|
||||
|
||||
@@ -181,7 +181,7 @@ Satisfying these requirements in Hindsight is straightforward. When new user inp
|
||||
|
||||

|
||||
|
||||
Most agent memory implementation rely on basic vector search or sometimes use a knowledge graph. Hindsight uses biomimetic data structures to organize agent memories in a way that is more like how human memory works:
|
||||
Most agent memory implementations rely on basic vector search or sometimes use a knowledge graph. Hindsight uses biomimetic data structures to organize agent memories in a way that is more like how human memory works:
|
||||
|
||||
- **World:** Facts about the world ("The stove gets hot")
|
||||
- **Experiences:** Agent's own experiences ("I touched the stove and it really hurt")
|
||||
@@ -307,3 +307,5 @@ MIT — see [LICENSE](./LICENSE)
|
||||
---
|
||||
|
||||
Built by [Vectorize.io](https://vectorize.io)
|
||||
|
||||
<img src="https://umami-pixel.chris-latimer.workers.dev/?id=a8b043e6-6964-454d-80df-69b69d3f0d50&host=github.com&url=/vectorize-io/hindsight" width="1" height="1" alt="" />
|
||||
|
||||
@@ -170,6 +170,11 @@ RUN chown -R hindsight:hindsight /app
|
||||
|
||||
USER hindsight
|
||||
|
||||
# Create pg0 data directory as hindsight user so that Docker seeds new named
|
||||
# volumes with correct ownership (UID 1000) on first use, avoiding the
|
||||
# "Permission denied" error when mounting a fresh root-owned volume.
|
||||
RUN mkdir -p /home/hindsight/.pg0
|
||||
|
||||
ENV PATH="/app/api/.venv/bin:${PATH}"
|
||||
|
||||
# Pre-download tiktoken encoding (ALWAYS - required for token counting even in air-gapped envs)
|
||||
@@ -321,6 +326,11 @@ RUN chown -R hindsight:hindsight /app
|
||||
|
||||
USER hindsight
|
||||
|
||||
# Create pg0 data directory as hindsight user so that Docker seeds new named
|
||||
# volumes with correct ownership (UID 1000) on first use, avoiding the
|
||||
# "Permission denied" error when mounting a fresh root-owned volume.
|
||||
RUN mkdir -p /home/hindsight/.pg0
|
||||
|
||||
ENV PATH="/app/api/.venv/bin:${PATH}"
|
||||
|
||||
# Pre-download tiktoken encoding (ALWAYS - required for token counting even in air-gapped envs)
|
||||
|
||||
@@ -97,7 +97,7 @@ fi
|
||||
if [ "$ENABLE_CP" = "true" ]; then
|
||||
echo "🎛️ Starting Control Plane..."
|
||||
cd /app/control-plane
|
||||
PORT=9999 node server.js &
|
||||
PORT="${HINDSIGHT_CP_PORT:-9999}" node server.js &
|
||||
CP_PID=$!
|
||||
PIDS+=($CP_PID)
|
||||
else
|
||||
@@ -110,7 +110,7 @@ echo "✅ Hindsight is running!"
|
||||
echo ""
|
||||
echo "📍 Access:"
|
||||
if [ "$ENABLE_CP" = "true" ]; then
|
||||
echo " Control Plane: http://localhost:9999"
|
||||
echo " Control Plane: http://localhost:${HINDSIGHT_CP_PORT:-9999}"
|
||||
fi
|
||||
if [ "$ENABLE_API" = "true" ]; then
|
||||
echo " API: http://localhost:8888"
|
||||
|
||||
+26
-10
@@ -13,9 +13,9 @@
|
||||
# target - Optional: 'cp-only' for control plane, otherwise assumes API image (default: api)
|
||||
#
|
||||
# Environment variables:
|
||||
# GROQ_API_KEY - Required for API/standalone images (LLM verification)
|
||||
# HINDSIGHT_API_LLM_PROVIDER - LLM provider (default: groq)
|
||||
# HINDSIGHT_API_LLM_MODEL - LLM model (default: llama-3.3-70b-versatile)
|
||||
# HINDSIGHT_API_LLM_API_KEY - Required for API/standalone images (LLM verification)
|
||||
# HINDSIGHT_API_LLM_PROVIDER - LLM provider (default: openai)
|
||||
# HINDSIGHT_API_LLM_MODEL - LLM model (default: gpt-4o-mini)
|
||||
# HINDSIGHT_API_EMBEDDINGS_PROVIDER - Embeddings provider (optional, for slim images: openai, cohere, tei)
|
||||
# HINDSIGHT_API_EMBEDDINGS_OPENAI_API_KEY - OpenAI API key for embeddings (optional)
|
||||
# HINDSIGHT_API_RERANKER_PROVIDER - Reranker provider (optional, for slim images: cohere, tei)
|
||||
@@ -34,7 +34,7 @@
|
||||
# ./docker/test-image.sh hindsight-control-plane:test cp-only
|
||||
#
|
||||
# # Test slim image with external providers
|
||||
# export GROQ_API_KEY=gsk_xxx
|
||||
# export HINDSIGHT_API_LLM_API_KEY=sk_xxx
|
||||
# export HINDSIGHT_API_EMBEDDINGS_PROVIDER=openai
|
||||
# export HINDSIGHT_API_EMBEDDINGS_OPENAI_API_KEY=sk-xxx
|
||||
# export HINDSIGHT_API_RERANKER_PROVIDER=cohere
|
||||
@@ -60,8 +60,8 @@ IMAGE="${1:-}"
|
||||
TARGET="${2:-api}"
|
||||
TIMEOUT="${SMOKE_TEST_TIMEOUT:-120}"
|
||||
CONTAINER_NAME="${SMOKE_TEST_CONTAINER_NAME:-hindsight-smoke-test}"
|
||||
LLM_PROVIDER="${HINDSIGHT_API_LLM_PROVIDER:-groq}"
|
||||
LLM_MODEL="${HINDSIGHT_API_LLM_MODEL:-llama-3.3-70b-versatile}"
|
||||
LLM_PROVIDER="${HINDSIGHT_API_LLM_PROVIDER:-openai}"
|
||||
LLM_MODEL="${HINDSIGHT_API_LLM_MODEL:-gpt-4o-mini}"
|
||||
|
||||
# Validate arguments
|
||||
if [ -z "$IMAGE" ]; then
|
||||
@@ -88,9 +88,9 @@ else
|
||||
fi
|
||||
|
||||
# Check for required environment variables
|
||||
if [ "$NEEDS_LLM" = true ] && [ -z "${GROQ_API_KEY:-}" ]; then
|
||||
echo -e "${RED}Error: GROQ_API_KEY environment variable is required for API/standalone images${NC}"
|
||||
echo "Set it with: export GROQ_API_KEY=your-api-key"
|
||||
if [ "$NEEDS_LLM" = true ] && [ "$LLM_PROVIDER" != "vertexai" ] && [ -z "${HINDSIGHT_API_LLM_API_KEY:-}" ]; then
|
||||
echo -e "${RED}Error: HINDSIGHT_API_LLM_API_KEY environment variable is required for API/standalone images${NC}"
|
||||
echo "Set it with: export HINDSIGHT_API_LLM_API_KEY=your-api-key"
|
||||
exit 2
|
||||
fi
|
||||
|
||||
@@ -123,9 +123,25 @@ else
|
||||
# Build docker run command with required and optional env vars
|
||||
DOCKER_CMD="docker run -d --name $CONTAINER_NAME"
|
||||
DOCKER_CMD="$DOCKER_CMD -e HINDSIGHT_API_LLM_PROVIDER=$LLM_PROVIDER"
|
||||
DOCKER_CMD="$DOCKER_CMD -e HINDSIGHT_API_LLM_API_KEY=${GROQ_API_KEY}"
|
||||
if [ -n "${HINDSIGHT_API_LLM_API_KEY:-}" ]; then
|
||||
DOCKER_CMD="$DOCKER_CMD -e HINDSIGHT_API_LLM_API_KEY=${HINDSIGHT_API_LLM_API_KEY}"
|
||||
fi
|
||||
DOCKER_CMD="$DOCKER_CMD -e HINDSIGHT_API_LLM_MODEL=$LLM_MODEL"
|
||||
|
||||
# Add Vertex AI config if provider is vertexai
|
||||
if [ "$LLM_PROVIDER" = "vertexai" ]; then
|
||||
if [ -n "${HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY:-}" ]; then
|
||||
DOCKER_CMD="$DOCKER_CMD -v ${HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY}:/tmp/gcp-credentials.json:ro"
|
||||
DOCKER_CMD="$DOCKER_CMD -e HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY=/tmp/gcp-credentials.json"
|
||||
fi
|
||||
if [ -n "${HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID:-}" ]; then
|
||||
DOCKER_CMD="$DOCKER_CMD -e HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID=${HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID}"
|
||||
fi
|
||||
if [ -n "${HINDSIGHT_API_LLM_VERTEXAI_REGION:-}" ]; then
|
||||
DOCKER_CMD="$DOCKER_CMD -e HINDSIGHT_API_LLM_VERTEXAI_REGION=${HINDSIGHT_API_LLM_VERTEXAI_REGION}"
|
||||
fi
|
||||
fi
|
||||
|
||||
# Add optional embeddings provider config
|
||||
if [ -n "${HINDSIGHT_API_EMBEDDINGS_PROVIDER:-}" ]; then
|
||||
DOCKER_CMD="$DOCKER_CMD -e HINDSIGHT_API_EMBEDDINGS_PROVIDER=${HINDSIGHT_API_EMBEDDINGS_PROVIDER}"
|
||||
|
||||
@@ -6,24 +6,17 @@
|
||||
# It expects API keys to be set in environment variables.
|
||||
#
|
||||
# Usage:
|
||||
# export GROQ_API_KEY=gsk_xxx
|
||||
# export OPENAI_API_KEY=sk-xxx
|
||||
# export COHERE_API_KEY=xxx
|
||||
# ./docker/test-slim-local.sh
|
||||
#
|
||||
# Or inline:
|
||||
# GROQ_API_KEY=gsk_xxx OPENAI_API_KEY=sk_xxx COHERE_API_KEY=xxx ./docker/test-slim-local.sh
|
||||
# OPENAI_API_KEY=sk_xxx COHERE_API_KEY=xxx ./docker/test-slim-local.sh
|
||||
#
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# Check for required API keys
|
||||
if [ -z "${GROQ_API_KEY:-}" ]; then
|
||||
echo "❌ Error: GROQ_API_KEY environment variable is required"
|
||||
echo "Set it with: export GROQ_API_KEY=gsk_xxx"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [ -z "${OPENAI_API_KEY:-}" ]; then
|
||||
echo "❌ Error: OPENAI_API_KEY environment variable is required"
|
||||
echo "Set it with: export OPENAI_API_KEY=sk-xxx"
|
||||
@@ -41,7 +34,10 @@ IMAGE="${1:-hindsight-slim:test}"
|
||||
echo "Testing image: $IMAGE"
|
||||
echo ""
|
||||
|
||||
# Set up external providers
|
||||
# Set up LLM and external providers
|
||||
export HINDSIGHT_API_LLM_PROVIDER=openai
|
||||
export HINDSIGHT_API_LLM_API_KEY=$OPENAI_API_KEY
|
||||
export HINDSIGHT_API_LLM_MODEL=gpt-4o-mini
|
||||
export HINDSIGHT_API_EMBEDDINGS_PROVIDER=openai
|
||||
export HINDSIGHT_API_EMBEDDINGS_OPENAI_API_KEY=$OPENAI_API_KEY
|
||||
export HINDSIGHT_API_RERANKER_PROVIDER=cohere
|
||||
|
||||
@@ -2,8 +2,8 @@ apiVersion: v2
|
||||
name: hindsight
|
||||
description: Hindsight helm chart
|
||||
type: application
|
||||
version: 0.4.11
|
||||
appVersion: "0.4.11"
|
||||
version: 0.4.14
|
||||
appVersion: "0.4.14"
|
||||
keywords:
|
||||
- ai
|
||||
- memory
|
||||
|
||||
@@ -46,4 +46,4 @@ __all__ = [
|
||||
"RemoteTEICrossEncoder",
|
||||
"LLMConfig",
|
||||
]
|
||||
__version__ = "0.4.11"
|
||||
__version__ = "0.4.14"
|
||||
|
||||
@@ -0,0 +1,88 @@
|
||||
"""Add text_signals column to memory_units for enriched BM25 indexing.
|
||||
|
||||
text_signals stores a denormalized space-separated string of entity names
|
||||
(and future signals) to improve full-text search recall without polluting
|
||||
the stored fact text.
|
||||
|
||||
- vchord: text_signals included in tokenize() at insert time
|
||||
- native: search_vector GENERATED column regenerated to include text_signals
|
||||
- pg_textsearch: no change (index only supports a single base column)
|
||||
|
||||
Revision ID: a2b3c4d5e6f7
|
||||
Revises: z1u2v3w4x5y6
|
||||
Create Date: 2026-02-28
|
||||
"""
|
||||
|
||||
import os
|
||||
from collections.abc import Sequence
|
||||
|
||||
from alembic import context, op
|
||||
|
||||
revision: str = "a2b3c4d5e6f7"
|
||||
down_revision: str | Sequence[str] | None = "aa2b3c4d5e6f"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
|
||||
def _get_schema_prefix() -> str:
|
||||
schema = context.config.get_main_option("target_schema")
|
||||
return f'"{schema}".' if schema else ""
|
||||
|
||||
|
||||
def _detect_text_search_extension() -> str:
|
||||
return os.getenv("HINDSIGHT_API_TEXT_SEARCH_EXTENSION", "native").lower()
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
schema = _get_schema_prefix()
|
||||
table = f"{schema}memory_units"
|
||||
text_search_ext = _detect_text_search_extension()
|
||||
|
||||
# Add text_signals column (nullable TEXT, populated at retain time)
|
||||
op.execute(f"ALTER TABLE {table} ADD COLUMN IF NOT EXISTS text_signals TEXT")
|
||||
|
||||
if text_search_ext == "native":
|
||||
# Native PostgreSQL: drop and recreate the GENERATED tsvector column to include text_signals
|
||||
op.execute(f"ALTER TABLE {table} DROP COLUMN IF EXISTS search_vector")
|
||||
op.execute(f"""
|
||||
ALTER TABLE {table}
|
||||
ADD COLUMN search_vector tsvector
|
||||
GENERATED ALWAYS AS (
|
||||
to_tsvector('english',
|
||||
COALESCE(text, '') || ' ' ||
|
||||
COALESCE(context, '') || ' ' ||
|
||||
COALESCE(text_signals, '')
|
||||
)
|
||||
) STORED
|
||||
""")
|
||||
# Recreate GIN index (was dropped with the column)
|
||||
op.execute(f"""
|
||||
CREATE INDEX IF NOT EXISTS idx_memory_units_text_search
|
||||
ON {table} USING gin(search_vector)
|
||||
""")
|
||||
|
||||
# vchord: tokenize() call in fact_storage.py is updated to include text_signals at insert time
|
||||
# pg_textsearch: no change — index operates on the base `text` column only
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
schema = _get_schema_prefix()
|
||||
table = f"{schema}memory_units"
|
||||
text_search_ext = _detect_text_search_extension()
|
||||
|
||||
if text_search_ext == "native":
|
||||
op.execute(f"DROP INDEX IF EXISTS {schema}idx_memory_units_text_search")
|
||||
op.execute(f"ALTER TABLE {table} DROP COLUMN IF EXISTS search_vector")
|
||||
op.execute(f"""
|
||||
ALTER TABLE {table}
|
||||
ADD COLUMN search_vector tsvector
|
||||
GENERATED ALWAYS AS (
|
||||
to_tsvector('english', COALESCE(text, '') || ' ' || COALESCE(context, ''))
|
||||
) STORED
|
||||
""")
|
||||
op.execute(f"""
|
||||
CREATE INDEX idx_memory_units_text_search
|
||||
ON {table} USING gin(search_vector)
|
||||
""")
|
||||
|
||||
op.execute(f"ALTER TABLE {table} DROP COLUMN IF EXISTS text_signals")
|
||||
@@ -0,0 +1,36 @@
|
||||
"""Make event_date nullable in memory_units to support timestamp-free content
|
||||
|
||||
Revision ID: aa2b3c4d5e6f
|
||||
Revises: z1u2v3w4x5y6
|
||||
Create Date: 2026-03-02
|
||||
|
||||
When callers retain content without a timestamp (e.g. fictional documents, static text),
|
||||
the event_date column should be allowed to be NULL rather than defaulting to utcnow().
|
||||
"""
|
||||
|
||||
from collections.abc import Sequence
|
||||
|
||||
from alembic import context, op
|
||||
|
||||
revision: str = "aa2b3c4d5e6f"
|
||||
down_revision: str | Sequence[str] | None = "z1u2v3w4x5y6"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
|
||||
def _get_schema_prefix() -> str:
|
||||
"""Get schema prefix for table names (required for multi-tenant support)."""
|
||||
schema = context.config.get_main_option("target_schema")
|
||||
return f'"{schema}".' if schema else ""
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
schema = _get_schema_prefix()
|
||||
op.execute(f"ALTER TABLE {schema}memory_units ALTER COLUMN event_date DROP NOT NULL")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
schema = _get_schema_prefix()
|
||||
# Backfill NULLs with now() before restoring the NOT NULL constraint
|
||||
op.execute(f"UPDATE {schema}memory_units SET event_date = now() WHERE event_date IS NULL")
|
||||
op.execute(f"ALTER TABLE {schema}memory_units ALTER COLUMN event_date SET NOT NULL")
|
||||
+34
@@ -0,0 +1,34 @@
|
||||
"""Backfill observation_scopes column if missing.
|
||||
|
||||
This migration ensures observation_scopes exists even on databases that had
|
||||
revision z1u2v3w4x5y6 applied when it referred to the old text_signals migration
|
||||
(before it was renamed to a2b3c4d5e6f7). The ADD COLUMN IF NOT EXISTS makes this
|
||||
a no-op on databases that already have the column.
|
||||
|
||||
Revision ID: b4c5d6e7f8a9
|
||||
Revises: a2b3c4d5e6f7
|
||||
Create Date: 2026-03-02
|
||||
"""
|
||||
|
||||
from collections.abc import Sequence
|
||||
|
||||
from alembic import context, op
|
||||
|
||||
revision: str = "b4c5d6e7f8a9"
|
||||
down_revision: str | Sequence[str] | None = "a2b3c4d5e6f7"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
|
||||
def _get_schema_prefix() -> str:
|
||||
schema = context.config.get_main_option("target_schema")
|
||||
return f'"{schema}".' if schema else ""
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
schema = _get_schema_prefix()
|
||||
op.execute(f"ALTER TABLE {schema}memory_units ADD COLUMN IF NOT EXISTS observation_scopes JSONB")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
pass # intentionally no-op — safe to leave the column in place
|
||||
+35
@@ -0,0 +1,35 @@
|
||||
"""Add observation_scopes column to memory_units table
|
||||
|
||||
Revision ID: z1u2v3w4x5y6
|
||||
Revises: a1b2c3d4e5f6
|
||||
Create Date: 2026-02-25
|
||||
|
||||
Adds observation_scopes JSONB column to memory_units to control how observations
|
||||
are scoped during consolidation. Accepts "per_tag", "combined", or an explicit
|
||||
list of tag-set lists for custom multi-pass consolidation.
|
||||
"""
|
||||
|
||||
from collections.abc import Sequence
|
||||
|
||||
from alembic import context, op
|
||||
|
||||
revision: str = "z1u2v3w4x5y6"
|
||||
down_revision: str | Sequence[str] | None = "a1b2c3d4e5f6"
|
||||
branch_labels: str | Sequence[str] | None = None
|
||||
depends_on: str | Sequence[str] | None = None
|
||||
|
||||
|
||||
def _get_schema_prefix() -> str:
|
||||
"""Get schema prefix for table names (required for multi-tenant support)."""
|
||||
schema = context.config.get_main_option("target_schema")
|
||||
return f'"{schema}".' if schema else ""
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
schema = _get_schema_prefix()
|
||||
op.execute(f"ALTER TABLE {schema}memory_units ADD COLUMN IF NOT EXISTS observation_scopes JSONB")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
schema = _get_schema_prefix()
|
||||
op.execute(f"ALTER TABLE {schema}memory_units DROP COLUMN IF EXISTS observation_scopes")
|
||||
@@ -74,7 +74,7 @@ from hindsight_api.config import get_config
|
||||
from hindsight_api.engine.db_utils import acquire_with_retry
|
||||
from hindsight_api.engine.memory_engine import Budget, _get_tiktoken_encoding, fq_table
|
||||
from hindsight_api.engine.reflect.observations import Observation
|
||||
from hindsight_api.engine.response_models import VALID_RECALL_FACT_TYPES, TokenUsage
|
||||
from hindsight_api.engine.response_models import VALID_RECALL_FACT_TYPES, MemoryFact, TokenUsage
|
||||
from hindsight_api.engine.search.tags import TagsMatch
|
||||
from hindsight_api.extensions import HttpExtension, OperationValidationError, load_extension
|
||||
from hindsight_api.metrics import create_metrics_collector, get_metrics_collector, initialize_metrics
|
||||
@@ -97,6 +97,12 @@ class ChunkIncludeOptions(BaseModel):
|
||||
max_tokens: int = Field(default=8192, description="Maximum tokens for chunks (chunks may be truncated)")
|
||||
|
||||
|
||||
class SourceFactsIncludeOptions(BaseModel):
|
||||
"""Options for including source facts for observation-type results."""
|
||||
|
||||
max_tokens: int = Field(default=4096, description="Maximum tokens for source facts")
|
||||
|
||||
|
||||
class IncludeOptions(BaseModel):
|
||||
"""Options for including additional data in recall results."""
|
||||
|
||||
@@ -107,6 +113,10 @@ class IncludeOptions(BaseModel):
|
||||
chunks: ChunkIncludeOptions | None = Field(
|
||||
default=None, description="Include raw chunks. Set to {} to enable, null to disable (default: disabled)."
|
||||
)
|
||||
source_facts: SourceFactsIncludeOptions | None = Field(
|
||||
default=None,
|
||||
description="Include source facts for observation-type results. Set to {} to enable, null to disable (default: disabled).",
|
||||
)
|
||||
|
||||
|
||||
class RecallRequest(BaseModel):
|
||||
@@ -189,6 +199,9 @@ class RecallResult(BaseModel):
|
||||
metadata: dict[str, str] | None = None # User-defined metadata
|
||||
chunk_id: str | None = None # Chunk this fact was extracted from
|
||||
tags: list[str] | None = None # Visibility scope tags
|
||||
source_fact_ids: list[str] | None = (
|
||||
None # IDs of source facts (observation type only, when source_facts is enabled)
|
||||
)
|
||||
|
||||
|
||||
class EntityObservationResponse(BaseModel):
|
||||
@@ -340,6 +353,9 @@ class RecallResponse(BaseModel):
|
||||
default=None, description="Entity states for entities mentioned in results"
|
||||
)
|
||||
chunks: dict[str, ChunkData] | None = Field(default=None, description="Chunks for facts, keyed by chunk_id")
|
||||
source_facts: dict[str, RecallResult] | None = Field(
|
||||
default=None, description="Source facts for observation-type results, keyed by fact ID"
|
||||
)
|
||||
|
||||
|
||||
class EntityInput(BaseModel):
|
||||
@@ -367,7 +383,15 @@ class MemoryItem(BaseModel):
|
||||
)
|
||||
|
||||
content: str
|
||||
timestamp: datetime | None = None
|
||||
timestamp: datetime | str | None = Field(
|
||||
default=None,
|
||||
description=(
|
||||
"When the content occurred. "
|
||||
"Accepts an ISO 8601 datetime string (e.g. '2024-01-15T10:30:00Z'), null/omitted (defaults to now), "
|
||||
"or the special string 'unset' to explicitly store without any timestamp "
|
||||
"(use this for timeless content such as fictional documents or static reference material)."
|
||||
),
|
||||
)
|
||||
context: str | None = None
|
||||
metadata: dict[str, str] | None = None
|
||||
document_id: str | None = Field(default=None, description="Optional document ID for this memory item.")
|
||||
@@ -379,6 +403,16 @@ class MemoryItem(BaseModel):
|
||||
default=None,
|
||||
description="Optional tags for visibility scoping. Memories with tags can be filtered during recall.",
|
||||
)
|
||||
observation_scopes: Literal["per_tag", "combined", "all_combinations"] | list[list[str]] | None = Field(
|
||||
default=None,
|
||||
title="ObservationScopes",
|
||||
description=(
|
||||
"How to scope observations during consolidation. "
|
||||
"'per_tag' runs one consolidation pass per individual tag, creating separate observations for each tag. "
|
||||
"'combined' (default) runs a single pass with all tags together. "
|
||||
"A list of tag lists runs one pass per inner list, giving full control over which combinations to use."
|
||||
),
|
||||
)
|
||||
|
||||
@field_validator("timestamp", mode="before")
|
||||
@classmethod
|
||||
@@ -388,12 +422,14 @@ class MemoryItem(BaseModel):
|
||||
if isinstance(v, datetime):
|
||||
return v
|
||||
if isinstance(v, str):
|
||||
if v.lower() == "unset":
|
||||
return "unset"
|
||||
try:
|
||||
# Try parsing as ISO format
|
||||
return datetime.fromisoformat(v.replace("Z", "+00:00"))
|
||||
except ValueError as e:
|
||||
raise ValueError(
|
||||
f"Invalid timestamp/event_date format: '{v}'. Expected ISO format like '2024-01-15T10:30:00' or '2024-01-15T10:30:00Z'"
|
||||
f"Invalid timestamp/event_date format: '{v}'. Expected ISO format like '2024-01-15T10:30:00' or '2024-01-15T10:30:00Z', or the special value 'unset' to store without a timestamp."
|
||||
) from e
|
||||
raise ValueError(f"timestamp must be a string or datetime, got {type(v).__name__}")
|
||||
|
||||
@@ -413,7 +449,6 @@ class RetainRequest(BaseModel):
|
||||
},
|
||||
],
|
||||
"async": False,
|
||||
"document_tags": ["user_a", "user_b"],
|
||||
}
|
||||
}
|
||||
)
|
||||
@@ -426,7 +461,8 @@ class RetainRequest(BaseModel):
|
||||
)
|
||||
document_tags: list[str] | None = Field(
|
||||
default=None,
|
||||
description="Tags applied to all items in this request. These are merged with any item-level tags.",
|
||||
description="Deprecated. Use item-level tags instead.",
|
||||
deprecated=True,
|
||||
)
|
||||
|
||||
|
||||
@@ -863,18 +899,103 @@ class CreateBankRequest(BaseModel):
|
||||
model_config = ConfigDict(
|
||||
json_schema_extra={
|
||||
"example": {
|
||||
"name": "Alice",
|
||||
"disposition": {"skepticism": 3, "literalism": 3, "empathy": 3},
|
||||
"mission": "I am a PM helping my engineering team stay organized",
|
||||
"retain_mission": "Always include technical decisions and architectural trade-offs. Ignore meeting logistics.",
|
||||
"observations_mission": "Observations are stable facts about people and projects. Always include preferences and skills.",
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
name: str | None = None
|
||||
disposition: DispositionTraits | None = None
|
||||
mission: str | None = Field(default=None, description="The agent's mission")
|
||||
# Deprecated: use mission instead
|
||||
background: str | None = Field(default=None, description="Deprecated: use mission instead")
|
||||
# Deprecated fields — kept for backwards compatibility only
|
||||
name: str | None = Field(default=None, description="Deprecated: display label only, not advertised")
|
||||
disposition: DispositionTraits | None = Field(
|
||||
default=None, description="Deprecated: use update_bank_config instead"
|
||||
)
|
||||
disposition_skepticism: int | None = Field(
|
||||
default=None, ge=1, le=5, description="Deprecated: use update_bank_config instead"
|
||||
)
|
||||
disposition_literalism: int | None = Field(
|
||||
default=None, ge=1, le=5, description="Deprecated: use update_bank_config instead"
|
||||
)
|
||||
disposition_empathy: int | None = Field(
|
||||
default=None, ge=1, le=5, description="Deprecated: use update_bank_config instead"
|
||||
)
|
||||
# Deprecated: use update_bank_config with reflect_mission instead
|
||||
mission: str | None = Field(
|
||||
default=None, description="Deprecated: use update_bank_config with reflect_mission instead"
|
||||
)
|
||||
# Deprecated alias for mission
|
||||
background: str | None = Field(
|
||||
default=None, description="Deprecated: use update_bank_config with reflect_mission instead"
|
||||
)
|
||||
|
||||
# Reflect configuration
|
||||
reflect_mission: str | None = Field(
|
||||
default=None,
|
||||
description="Mission/context for Reflect operations. Guides how Reflect interprets and uses memories.",
|
||||
)
|
||||
|
||||
# Operational configuration (applied via config resolver)
|
||||
retain_mission: str | None = Field(
|
||||
default=None,
|
||||
description="Steers what gets extracted during retain(). Injected alongside built-in extraction rules.",
|
||||
)
|
||||
retain_extraction_mode: str | None = Field(
|
||||
default=None,
|
||||
description="Fact extraction mode: 'concise' (default), 'verbose', or 'custom'.",
|
||||
)
|
||||
retain_custom_instructions: str | None = Field(
|
||||
default=None,
|
||||
description="Custom extraction prompt. Only active when retain_extraction_mode is 'custom'.",
|
||||
)
|
||||
retain_chunk_size: int | None = Field(
|
||||
default=None,
|
||||
description="Maximum token size for each content chunk during retain.",
|
||||
)
|
||||
enable_observations: bool | None = Field(
|
||||
default=None,
|
||||
description="Toggle automatic observation consolidation after retain().",
|
||||
)
|
||||
observations_mission: str | None = Field(
|
||||
default=None,
|
||||
description="Controls what gets synthesised into observations. Replaces built-in consolidation rules entirely.",
|
||||
)
|
||||
|
||||
def get_config_updates(self) -> dict[str, Any]:
|
||||
"""Return only the config fields that were explicitly set.
|
||||
|
||||
reflect_mission takes precedence over deprecated mission/background aliases.
|
||||
Individual disposition_* fields take priority over the deprecated disposition dict.
|
||||
"""
|
||||
updates: dict[str, Any] = {}
|
||||
# Resolve reflect mission: reflect_mission (new) > mission (deprecated) > background (deprecated)
|
||||
resolved_reflect_mission = self.reflect_mission or self.mission or self.background
|
||||
if resolved_reflect_mission is not None:
|
||||
updates["reflect_mission"] = resolved_reflect_mission
|
||||
# Disposition: individual fields take priority over legacy disposition dict
|
||||
if self.disposition_skepticism is not None:
|
||||
updates["disposition_skepticism"] = self.disposition_skepticism
|
||||
elif self.disposition is not None:
|
||||
updates["disposition_skepticism"] = self.disposition.skepticism
|
||||
if self.disposition_literalism is not None:
|
||||
updates["disposition_literalism"] = self.disposition_literalism
|
||||
elif self.disposition is not None:
|
||||
updates["disposition_literalism"] = self.disposition.literalism
|
||||
if self.disposition_empathy is not None:
|
||||
updates["disposition_empathy"] = self.disposition_empathy
|
||||
elif self.disposition is not None:
|
||||
updates["disposition_empathy"] = self.disposition.empathy
|
||||
for field_name in (
|
||||
"retain_mission",
|
||||
"retain_extraction_mode",
|
||||
"retain_custom_instructions",
|
||||
"retain_chunk_size",
|
||||
"enable_observations",
|
||||
"observations_mission",
|
||||
):
|
||||
value = getattr(self, field_name)
|
||||
if value is not None:
|
||||
updates[field_name] = value
|
||||
return updates
|
||||
|
||||
|
||||
class BankConfigUpdate(BaseModel):
|
||||
@@ -1134,6 +1255,14 @@ class DeleteResponse(BaseModel):
|
||||
deleted_count: int | None = None
|
||||
|
||||
|
||||
class ClearMemoryObservationsResponse(BaseModel):
|
||||
"""Response model for clearing observations for a specific memory."""
|
||||
|
||||
model_config = ConfigDict(json_schema_extra={"example": {"deleted_count": 3}})
|
||||
|
||||
deleted_count: int
|
||||
|
||||
|
||||
class BankStatsResponse(BaseModel):
|
||||
"""Response model for bank statistics endpoint."""
|
||||
|
||||
@@ -1813,12 +1942,19 @@ def _register_routes(app: FastAPI):
|
||||
bank_id: str,
|
||||
type: str | None = None,
|
||||
limit: int = 1000,
|
||||
q: str | None = None,
|
||||
tags: list[str] | None = Query(None),
|
||||
tags_match: str = "all_strict",
|
||||
request_context: RequestContext = Depends(get_request_context),
|
||||
):
|
||||
"""Get graph data from database, filtered by bank_id and optionally by type."""
|
||||
try:
|
||||
data = await app.state.memory.get_graph_data(bank_id, type, limit=limit, request_context=request_context)
|
||||
data = await app.state.memory.get_graph_data(
|
||||
bank_id, type, limit=limit, q=q, tags=tags, tags_match=tags_match, request_context=request_context
|
||||
)
|
||||
return data
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -1867,6 +2003,8 @@ def _register_routes(app: FastAPI):
|
||||
request_context=request_context,
|
||||
)
|
||||
return data
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -1898,6 +2036,8 @@ def _register_routes(app: FastAPI):
|
||||
if data is None:
|
||||
raise HTTPException(status_code=404, detail=f"Memory unit '{memory_id}' not found")
|
||||
return data
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -1959,6 +2099,10 @@ def _register_routes(app: FastAPI):
|
||||
include_chunks = request.include.chunks is not None
|
||||
max_chunk_tokens = request.include.chunks.max_tokens if include_chunks else 8192
|
||||
|
||||
# Determine source facts inclusion settings
|
||||
include_source_facts = request.include.source_facts is not None
|
||||
max_source_facts_tokens = request.include.source_facts.max_tokens if include_source_facts else 4096
|
||||
|
||||
pre_recall = time.time() - handler_start
|
||||
# Run recall with tracing (record metrics)
|
||||
with metrics.record_operation(
|
||||
@@ -1977,14 +2121,16 @@ def _register_routes(app: FastAPI):
|
||||
max_entity_tokens=max_entity_tokens,
|
||||
include_chunks=include_chunks,
|
||||
max_chunk_tokens=max_chunk_tokens,
|
||||
include_source_facts=include_source_facts,
|
||||
max_source_facts_tokens=max_source_facts_tokens,
|
||||
request_context=request_context,
|
||||
tags=request.tags,
|
||||
tags_match=request.tags_match,
|
||||
)
|
||||
|
||||
# Convert core MemoryFact objects to API RecallResult objects (excluding internal metrics)
|
||||
recall_results = [
|
||||
RecallResult(
|
||||
def _fact_to_result(fact: "MemoryFact") -> RecallResult:
|
||||
return RecallResult(
|
||||
id=fact.id,
|
||||
text=fact.text,
|
||||
type=fact.fact_type,
|
||||
@@ -1996,9 +2142,10 @@ def _register_routes(app: FastAPI):
|
||||
document_id=fact.document_id,
|
||||
chunk_id=fact.chunk_id,
|
||||
tags=fact.tags,
|
||||
source_fact_ids=fact.source_fact_ids,
|
||||
)
|
||||
for fact in core_result.results
|
||||
]
|
||||
|
||||
recall_results = [_fact_to_result(fact) for fact in core_result.results]
|
||||
|
||||
# Convert chunks from engine to HTTP API format
|
||||
chunks_response = None
|
||||
@@ -2026,11 +2173,19 @@ def _register_routes(app: FastAPI):
|
||||
],
|
||||
)
|
||||
|
||||
# Convert source facts dict to API format
|
||||
source_facts_response = None
|
||||
if core_result.source_facts:
|
||||
source_facts_response = {
|
||||
fact_id: _fact_to_result(fact) for fact_id, fact in core_result.source_facts.items()
|
||||
}
|
||||
|
||||
response = RecallResponse(
|
||||
results=recall_results,
|
||||
trace=core_result.trace,
|
||||
entities=entities_response,
|
||||
chunks=chunks_response,
|
||||
source_facts=source_facts_response,
|
||||
)
|
||||
|
||||
handler_duration = time.time() - handler_start
|
||||
@@ -2226,6 +2381,13 @@ def _register_routes(app: FastAPI):
|
||||
try:
|
||||
# Authenticate and set tenant schema
|
||||
await app.state.memory._authenticate_tenant(request_context)
|
||||
if app.state.memory._operation_validator:
|
||||
from hindsight_api.extensions import BankReadContext
|
||||
|
||||
ctx = BankReadContext(bank_id=bank_id, operation="get_bank_stats", request_context=request_context)
|
||||
await app.state.memory._validate_operation(
|
||||
app.state.memory._operation_validator.validate_bank_read(ctx)
|
||||
)
|
||||
pool = await app.state.memory._get_pool()
|
||||
async with acquire_with_retry(pool) as conn:
|
||||
# Get node counts by fact_type
|
||||
@@ -2359,6 +2521,8 @@ def _register_routes(app: FastAPI):
|
||||
total_observations=total_observations,
|
||||
)
|
||||
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -2393,6 +2557,8 @@ def _register_routes(app: FastAPI):
|
||||
limit=data["limit"],
|
||||
offset=data["offset"],
|
||||
)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -2432,6 +2598,8 @@ def _register_routes(app: FastAPI):
|
||||
for obs in entity["observations"]
|
||||
],
|
||||
)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -2493,6 +2661,8 @@ def _register_routes(app: FastAPI):
|
||||
request_context=request_context,
|
||||
)
|
||||
return MentalModelListResponse(items=[MentalModelResponse(**m) for m in mental_models])
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -2651,6 +2821,8 @@ def _register_routes(app: FastAPI):
|
||||
if mental_model is None:
|
||||
raise HTTPException(status_code=404, detail=f"Mental model '{mental_model_id}' not found")
|
||||
return MentalModelResponse(**mental_model)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -2682,6 +2854,8 @@ def _register_routes(app: FastAPI):
|
||||
if not deleted:
|
||||
raise HTTPException(status_code=404, detail=f"Mental model '{mental_model_id}' not found")
|
||||
return {"status": "deleted"}
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -2724,6 +2898,8 @@ def _register_routes(app: FastAPI):
|
||||
request_context=request_context,
|
||||
)
|
||||
return DirectiveListResponse(items=[DirectiveResponse(**d) for d in directives])
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -2756,6 +2932,8 @@ def _register_routes(app: FastAPI):
|
||||
if directive is None:
|
||||
raise HTTPException(status_code=404, detail=f"Directive '{directive_id}' not found")
|
||||
return DirectiveResponse(**directive)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -2792,6 +2970,8 @@ def _register_routes(app: FastAPI):
|
||||
return DirectiveResponse(**directive)
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -2830,6 +3010,8 @@ def _register_routes(app: FastAPI):
|
||||
if directive is None:
|
||||
raise HTTPException(status_code=404, detail=f"Directive '{directive_id}' not found")
|
||||
return DirectiveResponse(**directive)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -2861,6 +3043,8 @@ def _register_routes(app: FastAPI):
|
||||
if not deleted:
|
||||
raise HTTPException(status_code=404, detail=f"Directive '{directive_id}' not found")
|
||||
return {"status": "deleted"}
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -2899,6 +3083,8 @@ def _register_routes(app: FastAPI):
|
||||
bank_id=bank_id, search_query=q, limit=limit, offset=offset, request_context=request_context
|
||||
)
|
||||
return data
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -2931,6 +3117,8 @@ def _register_routes(app: FastAPI):
|
||||
if not document:
|
||||
raise HTTPException(status_code=404, detail="Document not found")
|
||||
return document
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -2984,6 +3172,8 @@ def _register_routes(app: FastAPI):
|
||||
request_context=request_context,
|
||||
)
|
||||
return data
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -3013,6 +3203,8 @@ def _register_routes(app: FastAPI):
|
||||
if not chunk:
|
||||
raise HTTPException(status_code=404, detail="Chunk not found")
|
||||
return chunk
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -3057,6 +3249,8 @@ def _register_routes(app: FastAPI):
|
||||
document_id=document_id,
|
||||
memory_units_deleted=result["memory_units_deleted"],
|
||||
)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -3093,6 +3287,8 @@ def _register_routes(app: FastAPI):
|
||||
offset=offset,
|
||||
operations=[OperationResponse(**op) for op in result["operations"]],
|
||||
)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -3124,6 +3320,8 @@ def _register_routes(app: FastAPI):
|
||||
|
||||
result = await app.state.memory.get_operation_status(bank_id, operation_id, request_context=request_context)
|
||||
return OperationStatusResponse(**result)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -3156,6 +3354,8 @@ def _register_routes(app: FastAPI):
|
||||
return CancelOperationResponse(**result)
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=404, detail=str(e))
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -3172,6 +3372,7 @@ def _register_routes(app: FastAPI):
|
||||
description="Get disposition traits and mission for a memory bank. Auto-creates agent with defaults if not exists.",
|
||||
operation_id="get_bank_profile",
|
||||
tags=["Banks"],
|
||||
deprecated=True,
|
||||
)
|
||||
async def api_get_bank_profile(bank_id: str, request_context: RequestContext = Depends(get_request_context)):
|
||||
"""Get memory bank profile (disposition + mission)."""
|
||||
@@ -3191,6 +3392,8 @@ def _register_routes(app: FastAPI):
|
||||
mission=mission,
|
||||
background=mission, # Backwards compat
|
||||
)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -3207,6 +3410,7 @@ def _register_routes(app: FastAPI):
|
||||
description="Update bank's disposition traits (skepticism, literalism, empathy)",
|
||||
operation_id="update_bank_disposition",
|
||||
tags=["Banks"],
|
||||
deprecated=True,
|
||||
)
|
||||
async def api_update_bank_disposition(
|
||||
bank_id: str, request: UpdateDispositionRequest, request_context: RequestContext = Depends(get_request_context)
|
||||
@@ -3233,6 +3437,8 @@ def _register_routes(app: FastAPI):
|
||||
mission=mission,
|
||||
background=mission, # Backwards compat
|
||||
)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -3261,6 +3467,8 @@ def _register_routes(app: FastAPI):
|
||||
)
|
||||
mission = result.get("mission") or ""
|
||||
return BackgroundResponse(mission=mission, background=mission)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -3286,21 +3494,18 @@ def _register_routes(app: FastAPI):
|
||||
# Ensure bank exists by getting profile (auto-creates with defaults)
|
||||
await app.state.memory.get_bank_profile(bank_id, request_context=request_context)
|
||||
|
||||
# Update name and/or mission if provided (support both mission and deprecated background)
|
||||
mission_value = request.mission or request.background
|
||||
if request.name is not None or mission_value is not None:
|
||||
# Update name if provided (stored in DB for display only, deprecated)
|
||||
if request.name is not None:
|
||||
await app.state.memory.update_bank(
|
||||
bank_id,
|
||||
name=request.name,
|
||||
mission=mission_value,
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
# Update disposition if provided
|
||||
if request.disposition is not None:
|
||||
await app.state.memory.update_bank_disposition(
|
||||
bank_id, request.disposition.model_dump(), request_context=request_context
|
||||
)
|
||||
# Apply all config overrides (includes reflect_mission, disposition, retain settings)
|
||||
config_updates = request.get_config_updates()
|
||||
if config_updates:
|
||||
await app.state.memory._config_resolver.update_bank_config(bank_id, config_updates, request_context)
|
||||
|
||||
# Get final profile
|
||||
final_profile = await app.state.memory.get_bank_profile(bank_id, request_context=request_context)
|
||||
@@ -3317,6 +3522,8 @@ def _register_routes(app: FastAPI):
|
||||
mission=mission,
|
||||
background=mission, # Backwards compat
|
||||
)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -3342,21 +3549,18 @@ def _register_routes(app: FastAPI):
|
||||
# Ensure bank exists
|
||||
await app.state.memory.get_bank_profile(bank_id, request_context=request_context)
|
||||
|
||||
# Update name and/or mission if provided
|
||||
mission_value = request.mission or request.background
|
||||
if request.name is not None or mission_value is not None:
|
||||
# Update name if provided (stored in DB for display only, deprecated)
|
||||
if request.name is not None:
|
||||
await app.state.memory.update_bank(
|
||||
bank_id,
|
||||
name=request.name,
|
||||
mission=mission_value,
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
# Update disposition if provided
|
||||
if request.disposition is not None:
|
||||
await app.state.memory.update_bank_disposition(
|
||||
bank_id, request.disposition.model_dump(), request_context=request_context
|
||||
)
|
||||
# Apply all config overrides (includes reflect_mission, disposition, retain settings)
|
||||
config_updates = request.get_config_updates()
|
||||
if config_updates:
|
||||
await app.state.memory._config_resolver.update_bank_config(bank_id, config_updates, request_context)
|
||||
|
||||
# Get final profile
|
||||
final_profile = await app.state.memory.get_bank_profile(bank_id, request_context=request_context)
|
||||
@@ -3373,6 +3577,8 @@ def _register_routes(app: FastAPI):
|
||||
mission=mission,
|
||||
background=mission, # Backwards compat
|
||||
)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -3402,6 +3608,8 @@ def _register_routes(app: FastAPI):
|
||||
+ result.get("entities_deleted", 0)
|
||||
+ result.get("documents_deleted", 0),
|
||||
)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -3428,6 +3636,8 @@ def _register_routes(app: FastAPI):
|
||||
message=f"Cleared {result.get('deleted_count', 0)} observations",
|
||||
deleted_count=result.get("deleted_count", 0),
|
||||
)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -3437,6 +3647,42 @@ def _register_routes(app: FastAPI):
|
||||
logger.error(f"Error in DELETE /v1/default/banks/{bank_id}/observations: {error_detail}")
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
@app.delete(
|
||||
"/v1/default/banks/{bank_id}/memories/{memory_id}/observations",
|
||||
response_model=ClearMemoryObservationsResponse,
|
||||
summary="Clear observations for a memory",
|
||||
description="Delete all observations derived from a specific memory and reset it for re-consolidation. "
|
||||
"The memory itself is not deleted. A consolidation job is triggered automatically so the memory "
|
||||
"will produce fresh observations on the next consolidation run.",
|
||||
operation_id="clear_memory_observations",
|
||||
tags=["Memory"],
|
||||
)
|
||||
async def api_clear_memory_observations(
|
||||
bank_id: str,
|
||||
memory_id: str,
|
||||
request_context: RequestContext = Depends(get_request_context),
|
||||
):
|
||||
"""Clear all observations derived from a specific memory."""
|
||||
try:
|
||||
result = await app.state.memory.clear_observations_for_memory(
|
||||
bank_id=bank_id,
|
||||
memory_id=memory_id,
|
||||
request_context=request_context,
|
||||
)
|
||||
return ClearMemoryObservationsResponse(deleted_count=result["deleted_count"])
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
import traceback
|
||||
|
||||
error_detail = f"{str(e)}\n\nTraceback:\n{traceback.format_exc()}"
|
||||
logger.error(
|
||||
f"Error in DELETE /v1/default/banks/{bank_id}/memories/{memory_id}/observations: {error_detail}"
|
||||
)
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
@app.get(
|
||||
"/v1/default/banks/{bank_id}/config",
|
||||
response_model=BankConfigResponse,
|
||||
@@ -3451,9 +3697,19 @@ def _register_routes(app: FastAPI):
|
||||
if not get_config().enable_bank_config_api:
|
||||
raise HTTPException(
|
||||
status_code=404,
|
||||
detail="Bank configuration API is disabled. Set HINDSIGHT_API_ENABLE_BANK_CONFIG_API=true to enable.",
|
||||
detail="Bank configuration API is disabled. Set HINDSIGHT_API_ENABLE_BANK_CONFIG_API=true to re-enable.",
|
||||
)
|
||||
try:
|
||||
# Authenticate and set schema context for multi-tenant DB queries
|
||||
await app.state.memory._authenticate_tenant(request_context)
|
||||
if app.state.memory._operation_validator:
|
||||
from hindsight_api.extensions import BankReadContext
|
||||
|
||||
ctx = BankReadContext(bank_id=bank_id, operation="get_bank_config", request_context=request_context)
|
||||
await app.state.memory._validate_operation(
|
||||
app.state.memory._operation_validator.validate_bank_read(ctx)
|
||||
)
|
||||
|
||||
# Get resolved config from config resolver
|
||||
config_dict = await app.state.memory._config_resolver.get_bank_config(bank_id, request_context)
|
||||
|
||||
@@ -3461,6 +3717,8 @@ def _register_routes(app: FastAPI):
|
||||
bank_overrides = await app.state.memory._config_resolver._load_bank_config(bank_id)
|
||||
|
||||
return BankConfigResponse(bank_id=bank_id, config=config_dict, overrides=bank_overrides)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -3486,9 +3744,19 @@ def _register_routes(app: FastAPI):
|
||||
if not get_config().enable_bank_config_api:
|
||||
raise HTTPException(
|
||||
status_code=404,
|
||||
detail="Bank configuration API is disabled. Set HINDSIGHT_API_ENABLE_BANK_CONFIG_API=true to enable.",
|
||||
detail="Bank configuration API is disabled. Set HINDSIGHT_API_ENABLE_BANK_CONFIG_API=true to re-enable.",
|
||||
)
|
||||
try:
|
||||
# Authenticate and set schema context for multi-tenant DB queries
|
||||
await app.state.memory._authenticate_tenant(request_context)
|
||||
if app.state.memory._operation_validator:
|
||||
from hindsight_api.extensions import BankWriteContext
|
||||
|
||||
ctx = BankWriteContext(bank_id=bank_id, operation="update_bank_config", request_context=request_context)
|
||||
await app.state.memory._validate_operation(
|
||||
app.state.memory._operation_validator.validate_bank_write(ctx)
|
||||
)
|
||||
|
||||
# Update config via config resolver (validates configurable fields and permissions)
|
||||
await app.state.memory._config_resolver.update_bank_config(bank_id, request.updates, request_context)
|
||||
|
||||
@@ -3497,6 +3765,8 @@ def _register_routes(app: FastAPI):
|
||||
bank_overrides = await app.state.memory._config_resolver._load_bank_config(bank_id)
|
||||
|
||||
return BankConfigResponse(bank_id=bank_id, config=config_dict, overrides=bank_overrides)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except ValueError as e:
|
||||
# Validation error (e.g., trying to override static field)
|
||||
raise HTTPException(status_code=400, detail=str(e))
|
||||
@@ -3523,9 +3793,19 @@ def _register_routes(app: FastAPI):
|
||||
if not get_config().enable_bank_config_api:
|
||||
raise HTTPException(
|
||||
status_code=404,
|
||||
detail="Bank configuration API is disabled. Set HINDSIGHT_API_ENABLE_BANK_CONFIG_API=true to enable.",
|
||||
detail="Bank configuration API is disabled. Set HINDSIGHT_API_ENABLE_BANK_CONFIG_API=true to re-enable.",
|
||||
)
|
||||
try:
|
||||
# Authenticate and set schema context for multi-tenant DB queries
|
||||
await app.state.memory._authenticate_tenant(request_context)
|
||||
if app.state.memory._operation_validator:
|
||||
from hindsight_api.extensions import BankWriteContext
|
||||
|
||||
ctx = BankWriteContext(bank_id=bank_id, operation="reset_bank_config", request_context=request_context)
|
||||
await app.state.memory._validate_operation(
|
||||
app.state.memory._operation_validator.validate_bank_write(ctx)
|
||||
)
|
||||
|
||||
# Reset config via config resolver
|
||||
await app.state.memory._config_resolver.reset_bank_config(bank_id)
|
||||
|
||||
@@ -3534,6 +3814,8 @@ def _register_routes(app: FastAPI):
|
||||
bank_overrides = await app.state.memory._config_resolver._load_bank_config(bank_id)
|
||||
|
||||
return BankConfigResponse(bank_id=bank_id, config=config_dict, overrides=bank_overrides)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -3559,6 +3841,8 @@ def _register_routes(app: FastAPI):
|
||||
operation_id=result["operation_id"],
|
||||
deduplicated=result.get("deduplicated", False),
|
||||
)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
@@ -3604,7 +3888,9 @@ def _register_routes(app: FastAPI):
|
||||
contents = []
|
||||
for item in request.items:
|
||||
content_dict = {"content": item.content}
|
||||
if item.timestamp:
|
||||
if item.timestamp == "unset":
|
||||
content_dict["event_date"] = None
|
||||
elif item.timestamp:
|
||||
content_dict["event_date"] = item.timestamp
|
||||
if item.context:
|
||||
content_dict["context"] = item.context
|
||||
@@ -3616,6 +3902,8 @@ def _register_routes(app: FastAPI):
|
||||
content_dict["entities"] = [{"text": e.text, "type": e.type or "CONCEPT"} for e in item.entities]
|
||||
if item.tags:
|
||||
content_dict["tags"] = item.tags
|
||||
if item.observation_scopes is not None:
|
||||
content_dict["observation_scopes"] = item.observation_scopes
|
||||
contents.append(content_dict)
|
||||
|
||||
if request.async_:
|
||||
@@ -3844,6 +4132,8 @@ def _register_routes(app: FastAPI):
|
||||
await app.state.memory.delete_bank(bank_id, fact_type=type, request_context=request_context)
|
||||
|
||||
return DeleteResponse(success=True)
|
||||
except OperationValidationError as e:
|
||||
raise HTTPException(status_code=e.status_code, detail=e.reason)
|
||||
except (AuthenticationError, HTTPException):
|
||||
raise
|
||||
except Exception as e:
|
||||
|
||||
@@ -8,12 +8,48 @@ from contextvars import ContextVar
|
||||
from fastmcp import FastMCP
|
||||
|
||||
from hindsight_api import MemoryEngine
|
||||
from hindsight_api.config import _get_raw_config
|
||||
from hindsight_api.engine.memory_engine import _current_schema
|
||||
from hindsight_api.extensions import MCPExtension, load_extension
|
||||
from hindsight_api.extensions.tenant import AuthenticationError
|
||||
from hindsight_api.mcp_tools import MCPToolsConfig, register_mcp_tools
|
||||
from hindsight_api.models import RequestContext
|
||||
|
||||
# All tools available in the system (explicit list — no wildcards)
|
||||
_ALL_TOOLS: frozenset[str] = frozenset(
|
||||
{
|
||||
"retain",
|
||||
"recall",
|
||||
"reflect",
|
||||
"list_banks",
|
||||
"create_bank",
|
||||
"list_mental_models",
|
||||
"get_mental_model",
|
||||
"create_mental_model",
|
||||
"update_mental_model",
|
||||
"delete_mental_model",
|
||||
"refresh_mental_model",
|
||||
"list_directives",
|
||||
"create_directive",
|
||||
"delete_directive",
|
||||
"list_memories",
|
||||
"get_memory",
|
||||
"delete_memory",
|
||||
"list_documents",
|
||||
"get_document",
|
||||
"delete_document",
|
||||
"list_operations",
|
||||
"get_operation",
|
||||
"cancel_operation",
|
||||
"list_tags",
|
||||
"get_bank",
|
||||
"get_bank_stats",
|
||||
"update_bank",
|
||||
"delete_bank",
|
||||
"clear_memories",
|
||||
}
|
||||
)
|
||||
|
||||
# Configure logging from HINDSIGHT_API_LOG_LEVEL environment variable
|
||||
_log_level_str = os.environ.get("HINDSIGHT_API_LOG_LEVEL", "info").lower()
|
||||
_log_level_map = {
|
||||
@@ -78,21 +114,15 @@ def create_mcp_server(memory: MemoryEngine, multi_bank: bool = True) -> FastMCP:
|
||||
If False, only expose bank-scoped tools without bank_id parameters.
|
||||
|
||||
Returns:
|
||||
Configured FastMCP server instance with stateless_http enabled
|
||||
Configured FastMCP server instance
|
||||
"""
|
||||
# Use stateless_http=True for Claude Code compatibility
|
||||
mcp = FastMCP("hindsight-mcp-server", stateless_http=True)
|
||||
mcp = FastMCP("hindsight-mcp-server")
|
||||
|
||||
# Configure and register tools using shared module
|
||||
config = MCPToolsConfig(
|
||||
bank_id_resolver=get_current_bank_id,
|
||||
api_key_resolver=get_current_api_key, # Propagate API key for tenant auth
|
||||
tenant_id_resolver=get_current_tenant_id, # Propagate tenant_id for usage metering
|
||||
api_key_id_resolver=get_current_api_key_id, # Propagate api_key_id for usage metering
|
||||
include_bank_id_param=multi_bank,
|
||||
tools=None
|
||||
if multi_bank
|
||||
else {
|
||||
global_config = _get_raw_config()
|
||||
|
||||
# Tools available for this mode (multi-bank exposes all tools; single-bank excludes bank-management tools)
|
||||
_SINGLE_BANK_TOOLS: frozenset[str] = frozenset(
|
||||
{
|
||||
"retain",
|
||||
"recall",
|
||||
"reflect",
|
||||
@@ -102,8 +132,40 @@ def create_mcp_server(memory: MemoryEngine, multi_bank: bool = True) -> FastMCP:
|
||||
"update_mental_model",
|
||||
"delete_mental_model",
|
||||
"refresh_mental_model",
|
||||
}, # Scoped tools for single-bank mode (excludes bank management: list_banks, create_bank)
|
||||
retain_fire_and_forget=False, # HTTP MCP supports sync/async modes
|
||||
"list_directives",
|
||||
"create_directive",
|
||||
"delete_directive",
|
||||
"list_memories",
|
||||
"get_memory",
|
||||
"delete_memory",
|
||||
"list_documents",
|
||||
"get_document",
|
||||
"delete_document",
|
||||
"list_operations",
|
||||
"get_operation",
|
||||
"cancel_operation",
|
||||
"list_tags",
|
||||
"get_bank",
|
||||
"update_bank",
|
||||
"delete_bank",
|
||||
"clear_memories",
|
||||
}
|
||||
)
|
||||
base_tools: frozenset[str] | None = None if multi_bank else _SINGLE_BANK_TOOLS
|
||||
|
||||
# Apply global mcp_enabled_tools filter (env-level allowlist)
|
||||
if global_config.mcp_enabled_tools is not None:
|
||||
allowed = frozenset(global_config.mcp_enabled_tools)
|
||||
base_tools = (base_tools if base_tools is not None else _ALL_TOOLS) & allowed
|
||||
|
||||
# Configure and register tools using shared module
|
||||
config = MCPToolsConfig(
|
||||
bank_id_resolver=get_current_bank_id,
|
||||
api_key_resolver=get_current_api_key, # Propagate API key for tenant auth
|
||||
tenant_id_resolver=get_current_tenant_id, # Propagate tenant_id for usage metering
|
||||
api_key_id_resolver=get_current_api_key_id, # Propagate api_key_id for usage metering
|
||||
include_bank_id_param=multi_bank,
|
||||
tools=base_tools,
|
||||
)
|
||||
|
||||
register_mcp_tools(mcp, memory, config)
|
||||
@@ -211,9 +273,9 @@ class MCPMiddleware:
|
||||
else:
|
||||
# Create servers internally (for direct construction / tests)
|
||||
self.multi_bank_server = create_mcp_server(memory, multi_bank=True)
|
||||
self.multi_bank_app = self.multi_bank_server.http_app(path="/")
|
||||
self.multi_bank_app = self.multi_bank_server.http_app(path="/", stateless_http=True)
|
||||
self.single_bank_server = create_mcp_server(memory, multi_bank=False)
|
||||
self.single_bank_app = self.single_bank_server.http_app(path="/")
|
||||
self.single_bank_app = self.single_bank_server.http_app(path="/", stateless_http=True)
|
||||
|
||||
def _get_header(self, scope: dict, name: str) -> str | None:
|
||||
"""Extract a header value from ASGI scope."""
|
||||
@@ -379,9 +441,9 @@ def create_mcp_servers(memory: MemoryEngine):
|
||||
Tuple of (multi_bank_server, single_bank_server, multi_bank_app, single_bank_app)
|
||||
"""
|
||||
multi_bank_server = create_mcp_server(memory, multi_bank=True)
|
||||
multi_bank_app = multi_bank_server.http_app(path="/")
|
||||
multi_bank_app = multi_bank_server.http_app(path="/", stateless_http=True)
|
||||
|
||||
single_bank_server = create_mcp_server(memory, multi_bank=False)
|
||||
single_bank_app = single_bank_server.http_app(path="/")
|
||||
single_bank_app = single_bank_server.http_app(path="/", stateless_http=True)
|
||||
|
||||
return multi_bank_server, single_bank_server, multi_bank_app, single_bank_app
|
||||
|
||||
@@ -218,6 +218,10 @@ ENV_RERANKER_MAX_CANDIDATES = "HINDSIGHT_API_RERANKER_MAX_CANDIDATES"
|
||||
ENV_RERANKER_FLASHRANK_MODEL = "HINDSIGHT_API_RERANKER_FLASHRANK_MODEL"
|
||||
ENV_RERANKER_FLASHRANK_CACHE_DIR = "HINDSIGHT_API_RERANKER_FLASHRANK_CACHE_DIR"
|
||||
|
||||
# ZeroEntropy configuration (reranker only)
|
||||
ENV_RERANKER_ZEROENTROPY_API_KEY = "HINDSIGHT_API_RERANKER_ZEROENTROPY_API_KEY"
|
||||
ENV_RERANKER_ZEROENTROPY_MODEL = "HINDSIGHT_API_RERANKER_ZEROENTROPY_MODEL"
|
||||
|
||||
ENV_VECTOR_EXTENSION = "HINDSIGHT_API_VECTOR_EXTENSION"
|
||||
ENV_TEXT_SEARCH_EXTENSION = "HINDSIGHT_API_TEXT_SEARCH_EXTENSION"
|
||||
|
||||
@@ -228,13 +232,12 @@ ENV_LOG_LEVEL = "HINDSIGHT_API_LOG_LEVEL"
|
||||
ENV_LOG_FORMAT = "HINDSIGHT_API_LOG_FORMAT"
|
||||
ENV_WORKERS = "HINDSIGHT_API_WORKERS"
|
||||
ENV_MCP_ENABLED = "HINDSIGHT_API_MCP_ENABLED"
|
||||
ENV_MCP_ENABLED_TOOLS = "HINDSIGHT_API_MCP_ENABLED_TOOLS"
|
||||
ENV_ENABLE_BANK_CONFIG_API = "HINDSIGHT_API_ENABLE_BANK_CONFIG_API"
|
||||
ENV_GRAPH_RETRIEVER = "HINDSIGHT_API_GRAPH_RETRIEVER"
|
||||
ENV_MPFP_TOP_K_NEIGHBORS = "HINDSIGHT_API_MPFP_TOP_K_NEIGHBORS"
|
||||
ENV_RECALL_MAX_CONCURRENT = "HINDSIGHT_API_RECALL_MAX_CONCURRENT"
|
||||
ENV_RECALL_CONNECTION_BUDGET = "HINDSIGHT_API_RECALL_CONNECTION_BUDGET"
|
||||
ENV_MCP_LOCAL_BANK_ID = "HINDSIGHT_API_MCP_LOCAL_BANK_ID"
|
||||
ENV_MCP_INSTRUCTIONS = "HINDSIGHT_API_MCP_INSTRUCTIONS"
|
||||
ENV_MENTAL_MODEL_REFRESH_CONCURRENCY = "HINDSIGHT_API_MENTAL_MODEL_REFRESH_CONCURRENCY"
|
||||
|
||||
# OpenTelemetry tracing configuration
|
||||
@@ -254,6 +257,7 @@ ENV_RETAIN_MAX_COMPLETION_TOKENS = "HINDSIGHT_API_RETAIN_MAX_COMPLETION_TOKENS"
|
||||
ENV_RETAIN_CHUNK_SIZE = "HINDSIGHT_API_RETAIN_CHUNK_SIZE"
|
||||
ENV_RETAIN_EXTRACT_CAUSAL_LINKS = "HINDSIGHT_API_RETAIN_EXTRACT_CAUSAL_LINKS"
|
||||
ENV_RETAIN_EXTRACTION_MODE = "HINDSIGHT_API_RETAIN_EXTRACTION_MODE"
|
||||
ENV_RETAIN_MISSION = "HINDSIGHT_API_RETAIN_MISSION"
|
||||
ENV_RETAIN_CUSTOM_INSTRUCTIONS = "HINDSIGHT_API_RETAIN_CUSTOM_INSTRUCTIONS"
|
||||
ENV_RETAIN_BATCH_TOKENS = "HINDSIGHT_API_RETAIN_BATCH_TOKENS"
|
||||
ENV_RETAIN_BATCH_ENABLED = "HINDSIGHT_API_RETAIN_BATCH_ENABLED"
|
||||
@@ -272,6 +276,8 @@ ENV_FILE_STORAGE_AZURE_CONTAINER = "HINDSIGHT_API_FILE_STORAGE_AZURE_CONTAINER"
|
||||
ENV_FILE_STORAGE_AZURE_ACCOUNT_NAME = "HINDSIGHT_API_FILE_STORAGE_AZURE_ACCOUNT_NAME"
|
||||
ENV_FILE_STORAGE_AZURE_ACCOUNT_KEY = "HINDSIGHT_API_FILE_STORAGE_AZURE_ACCOUNT_KEY"
|
||||
ENV_FILE_PARSER = "HINDSIGHT_API_FILE_PARSER"
|
||||
ENV_FILE_PARSER_IRIS_TOKEN = "HINDSIGHT_API_FILE_PARSER_IRIS_TOKEN"
|
||||
ENV_FILE_PARSER_IRIS_ORG_ID = "HINDSIGHT_API_FILE_PARSER_IRIS_ORG_ID"
|
||||
ENV_FILE_CONVERSION_MAX_BATCH_SIZE_MB = "HINDSIGHT_API_FILE_CONVERSION_MAX_BATCH_SIZE_MB"
|
||||
ENV_FILE_CONVERSION_MAX_BATCH_SIZE = "HINDSIGHT_API_FILE_CONVERSION_MAX_BATCH_SIZE"
|
||||
ENV_ENABLE_FILE_UPLOAD_API = "HINDSIGHT_API_ENABLE_FILE_UPLOAD_API"
|
||||
@@ -280,7 +286,9 @@ ENV_FILE_DELETE_AFTER_RETAIN = "HINDSIGHT_API_FILE_DELETE_AFTER_RETAIN"
|
||||
# Observations settings (consolidated knowledge from facts)
|
||||
ENV_ENABLE_OBSERVATIONS = "HINDSIGHT_API_ENABLE_OBSERVATIONS"
|
||||
ENV_CONSOLIDATION_BATCH_SIZE = "HINDSIGHT_API_CONSOLIDATION_BATCH_SIZE"
|
||||
ENV_CONSOLIDATION_LLM_BATCH_SIZE = "HINDSIGHT_API_CONSOLIDATION_LLM_BATCH_SIZE"
|
||||
ENV_CONSOLIDATION_MAX_TOKENS = "HINDSIGHT_API_CONSOLIDATION_MAX_TOKENS"
|
||||
ENV_OBSERVATIONS_MISSION = "HINDSIGHT_API_OBSERVATIONS_MISSION"
|
||||
|
||||
# Optimization flags
|
||||
ENV_SKIP_LLM_VERIFICATION = "HINDSIGHT_API_SKIP_LLM_VERIFICATION"
|
||||
@@ -306,6 +314,13 @@ ENV_WORKER_CONSOLIDATION_MAX_SLOTS = "HINDSIGHT_API_WORKER_CONSOLIDATION_MAX_SLO
|
||||
|
||||
# Reflect agent settings
|
||||
ENV_REFLECT_MAX_ITERATIONS = "HINDSIGHT_API_REFLECT_MAX_ITERATIONS"
|
||||
ENV_REFLECT_MAX_CONTEXT_TOKENS = "HINDSIGHT_API_REFLECT_MAX_CONTEXT_TOKENS"
|
||||
ENV_REFLECT_MISSION = "HINDSIGHT_API_REFLECT_MISSION"
|
||||
|
||||
# Disposition settings
|
||||
ENV_DISPOSITION_SKEPTICISM = "HINDSIGHT_API_DISPOSITION_SKEPTICISM"
|
||||
ENV_DISPOSITION_LITERALISM = "HINDSIGHT_API_DISPOSITION_LITERALISM"
|
||||
ENV_DISPOSITION_EMPATHY = "HINDSIGHT_API_DISPOSITION_EMPATHY"
|
||||
|
||||
# Default values
|
||||
DEFAULT_DATABASE_URL = "pg0"
|
||||
@@ -314,18 +329,18 @@ DEFAULT_LLM_PROVIDER = "openai"
|
||||
|
||||
# Provider-specific default models
|
||||
PROVIDER_DEFAULT_MODELS = {
|
||||
"openai": "o3-mini",
|
||||
"openai": "gpt-4o-mini",
|
||||
"anthropic": "claude-haiku-4-5-20251001",
|
||||
"gemini": "gemini-2.5-flash",
|
||||
"groq": "openai/gpt-oss-120b",
|
||||
"ollama": "gemma3:12b",
|
||||
"lmstudio": "local-model",
|
||||
"vertexai": "gemini-2.0-flash-001",
|
||||
"vertexai": "google/gemini-2.5-flash-lite",
|
||||
"openai-codex": "gpt-5.2-codex",
|
||||
"claude-code": "claude-sonnet-4-5-20250929",
|
||||
"mock": "mock-model",
|
||||
}
|
||||
DEFAULT_LLM_MODEL = "o3-mini" # Fallback if provider not in table
|
||||
DEFAULT_LLM_MODEL = "gpt-4o-mini" # Fallback if provider not in table
|
||||
DEFAULT_LLM_MAX_CONCURRENT = 32
|
||||
DEFAULT_LLM_MAX_RETRIES = 10 # Max retry attempts for LLM API calls
|
||||
DEFAULT_LLM_INITIAL_BACKOFF = 1.0 # Initial backoff in seconds for retry exponential backoff
|
||||
@@ -360,6 +375,8 @@ DEFAULT_RERANKER_FLASHRANK_CACHE_DIR = None # Use default cache directory
|
||||
DEFAULT_EMBEDDINGS_COHERE_MODEL = "embed-english-v3.0"
|
||||
DEFAULT_RERANKER_COHERE_MODEL = "rerank-english-v3.0"
|
||||
|
||||
DEFAULT_RERANKER_ZEROENTROPY_MODEL = "zerank-2"
|
||||
|
||||
# Vector extension (pgvector, vchord, or pgvectorscale)
|
||||
DEFAULT_VECTOR_EXTENSION = "pgvector" # Options: "pgvector", "vchord", "pgvectorscale"
|
||||
|
||||
@@ -382,12 +399,12 @@ DEFAULT_LOG_LEVEL = "info"
|
||||
DEFAULT_LOG_FORMAT = "text" # Options: "text", "json"
|
||||
DEFAULT_WORKERS = 1
|
||||
DEFAULT_MCP_ENABLED = True
|
||||
DEFAULT_ENABLE_BANK_CONFIG_API = False # Disabled by default for security
|
||||
DEFAULT_MCP_ENABLED_TOOLS: list[str] | None = None # None = all tools enabled
|
||||
DEFAULT_ENABLE_BANK_CONFIG_API = True
|
||||
DEFAULT_GRAPH_RETRIEVER = "link_expansion" # Options: "link_expansion", "mpfp", "bfs"
|
||||
DEFAULT_MPFP_TOP_K_NEIGHBORS = 20 # Fan-out limit per node in MPFP graph traversal
|
||||
DEFAULT_RECALL_MAX_CONCURRENT = 32 # Max concurrent recall operations per worker
|
||||
DEFAULT_RECALL_CONNECTION_BUDGET = 4 # Max concurrent DB connections per recall operation
|
||||
DEFAULT_MCP_LOCAL_BANK_ID = "mcp"
|
||||
DEFAULT_MENTAL_MODEL_REFRESH_CONCURRENCY = 8 # Max concurrent mental model refreshes
|
||||
|
||||
# Retain settings
|
||||
@@ -396,6 +413,7 @@ DEFAULT_RETAIN_CHUNK_SIZE = 3000 # Max chars per chunk for fact extraction
|
||||
DEFAULT_RETAIN_EXTRACT_CAUSAL_LINKS = True # Extract causal links between facts
|
||||
DEFAULT_RETAIN_EXTRACTION_MODE = "concise" # Extraction mode: "concise", "verbose", or "custom"
|
||||
RETAIN_EXTRACTION_MODES = ("concise", "verbose", "custom") # Allowed extraction modes
|
||||
DEFAULT_RETAIN_MISSION = None # Declarative spec of what to retain (injected into any extraction mode)
|
||||
DEFAULT_RETAIN_CUSTOM_INSTRUCTIONS = None # Custom extraction guidelines (only used when mode="custom")
|
||||
DEFAULT_RETAIN_BATCH_TOKENS = 10_000 # ~40KB of text # Max chars per sub-batch for async retain auto-splitting
|
||||
DEFAULT_RETAIN_BATCH_ENABLED = False # Use LLM Batch API for fact extraction (only when async=True)
|
||||
@@ -412,7 +430,9 @@ DEFAULT_FILE_DELETE_AFTER_RETAIN = True # Delete file bytes after retain (saves
|
||||
# Observations defaults (consolidated knowledge from facts)
|
||||
DEFAULT_ENABLE_OBSERVATIONS = True # Observations enabled by default
|
||||
DEFAULT_CONSOLIDATION_BATCH_SIZE = 50 # Memories to load per batch (internal memory optimization)
|
||||
DEFAULT_CONSOLIDATION_MAX_TOKENS = 1024 # Max tokens for recall when finding related observations
|
||||
DEFAULT_CONSOLIDATION_LLM_BATCH_SIZE = 8 # Facts per LLM call (1 = no batching; >1 = batch mode)
|
||||
DEFAULT_CONSOLIDATION_MAX_TOKENS = 512 # Max tokens for recall when finding related observations
|
||||
DEFAULT_OBSERVATIONS_MISSION = None # Declarative spec of what observations are for this bank
|
||||
|
||||
# Database migrations
|
||||
DEFAULT_RUN_MIGRATIONS_ON_STARTUP = True
|
||||
@@ -434,6 +454,12 @@ DEFAULT_WORKER_CONSOLIDATION_MAX_SLOTS = 2 # Max concurrent consolidation tasks
|
||||
|
||||
# Reflect agent settings
|
||||
DEFAULT_REFLECT_MAX_ITERATIONS = 10 # Max tool call iterations before forcing response
|
||||
DEFAULT_REFLECT_MAX_CONTEXT_TOKENS = 100_000 # Max accumulated context tokens before forcing final prompt
|
||||
|
||||
# Disposition defaults (None = not set, fall back to bank DB value or 3)
|
||||
DEFAULT_DISPOSITION_SKEPTICISM = None
|
||||
DEFAULT_DISPOSITION_LITERALISM = None
|
||||
DEFAULT_DISPOSITION_EMPATHY = None
|
||||
|
||||
# OpenTelemetry tracing configuration
|
||||
DEFAULT_OTEL_TRACES_ENABLED = False # Disabled by default for backward compatibility
|
||||
@@ -606,6 +632,8 @@ class HindsightConfig:
|
||||
reranker_litellm_sdk_api_key: str | None
|
||||
reranker_litellm_sdk_model: str
|
||||
reranker_litellm_sdk_api_base: str | None
|
||||
reranker_zeroentropy_api_key: str | None
|
||||
reranker_zeroentropy_model: str
|
||||
|
||||
# Server
|
||||
host: str
|
||||
@@ -614,6 +642,7 @@ class HindsightConfig:
|
||||
log_level: str
|
||||
log_format: str
|
||||
mcp_enabled: bool
|
||||
mcp_enabled_tools: list[str] | None # None = all tools; explicit list = allowlist
|
||||
enable_bank_config_api: bool
|
||||
|
||||
# Recall
|
||||
@@ -628,6 +657,7 @@ class HindsightConfig:
|
||||
retain_chunk_size: int
|
||||
retain_extract_causal_links: bool
|
||||
retain_extraction_mode: str
|
||||
retain_mission: str | None
|
||||
retain_custom_instructions: str | None
|
||||
retain_batch_tokens: int
|
||||
retain_batch_enabled: bool
|
||||
@@ -645,7 +675,9 @@ class HindsightConfig:
|
||||
file_storage_azure_container: str | None # Azure container name (required for azure storage)
|
||||
file_storage_azure_account_name: str | None # Azure storage account name
|
||||
file_storage_azure_account_key: str | None # Azure storage account key
|
||||
file_parser: str # File parser to use (e.g., "markitdown")
|
||||
file_parser: str # File parser to use (e.g., "markitdown", "iris")
|
||||
file_parser_iris_token: str | None # Vectorize API token for iris parser (VECTORIZE_TOKEN)
|
||||
file_parser_iris_org_id: str | None # Vectorize org ID for iris parser (VECTORIZE_ORG_ID)
|
||||
file_conversion_max_batch_size_mb: int # Max total batch size in MB (all files combined)
|
||||
file_conversion_max_batch_size: int # Max files per request
|
||||
enable_file_upload_api: bool
|
||||
@@ -654,7 +686,24 @@ class HindsightConfig:
|
||||
# Observations settings (consolidated knowledge from facts)
|
||||
enable_observations: bool
|
||||
consolidation_batch_size: int
|
||||
consolidation_llm_batch_size: int
|
||||
consolidation_max_tokens: int
|
||||
observations_mission: str | None
|
||||
|
||||
# Entity labels (controlled vocabulary of key:value classification labels extracted at retain time)
|
||||
# List of label group dicts: [{key, description, type, optional, values: [{value, description}]}]
|
||||
entity_labels: list | None
|
||||
# Whether to extract regular named entities alongside entity labels (default: True)
|
||||
# When False: only label entities are extracted (or no entities at all if no labels configured)
|
||||
entities_allow_free_form: bool
|
||||
|
||||
# Reflect agent settings
|
||||
reflect_mission: str | None
|
||||
|
||||
# Disposition settings (hierarchical - can be overridden per bank; None = fall back to DB)
|
||||
disposition_skepticism: int | None
|
||||
disposition_literalism: int | None
|
||||
disposition_empathy: int | None
|
||||
|
||||
# Optimization flags
|
||||
skip_llm_verification: bool
|
||||
@@ -680,6 +729,7 @@ class HindsightConfig:
|
||||
|
||||
# Reflect agent settings
|
||||
reflect_max_iterations: int
|
||||
reflect_max_context_tokens: int
|
||||
|
||||
# OpenTelemetry tracing configuration
|
||||
otel_traces_enabled: bool
|
||||
@@ -712,18 +762,33 @@ class HindsightConfig:
|
||||
"file_storage_s3_secret_access_key",
|
||||
"file_storage_gcs_service_account_key",
|
||||
"file_storage_azure_account_key",
|
||||
# File parser credentials
|
||||
"file_parser_iris_token",
|
||||
}
|
||||
|
||||
# CONFIGURABLE_FIELDS: Safe behavioral settings that can be customized per-tenant/bank
|
||||
# These fields are manually tagged as safe to expose and modify.
|
||||
# Excludes credentials, infrastructure config, provider/model selection, and performance tuning.
|
||||
_CONFIGURABLE_FIELDS = {
|
||||
# MCP tool access control
|
||||
"mcp_enabled_tools",
|
||||
# Retention settings (behavioral)
|
||||
"retain_chunk_size",
|
||||
"retain_extraction_mode",
|
||||
"retain_mission",
|
||||
"retain_custom_instructions",
|
||||
# Entity labels (controlled vocabulary for entity classification)
|
||||
"entity_labels",
|
||||
"entities_allow_free_form",
|
||||
# Consolidation settings
|
||||
"enable_observations",
|
||||
"observations_mission",
|
||||
# Reflect settings
|
||||
"reflect_mission",
|
||||
# Disposition settings
|
||||
"disposition_skepticism",
|
||||
"disposition_literalism",
|
||||
"disposition_empathy",
|
||||
}
|
||||
|
||||
@property
|
||||
@@ -976,6 +1041,9 @@ class HindsightConfig:
|
||||
reranker_litellm_sdk_api_key=os.getenv(ENV_RERANKER_LITELLM_SDK_API_KEY),
|
||||
reranker_litellm_sdk_model=os.getenv(ENV_RERANKER_LITELLM_SDK_MODEL, DEFAULT_RERANKER_LITELLM_SDK_MODEL),
|
||||
reranker_litellm_sdk_api_base=os.getenv(ENV_RERANKER_LITELLM_SDK_API_BASE) or None,
|
||||
# ZeroEntropy reranker
|
||||
reranker_zeroentropy_api_key=os.getenv(ENV_RERANKER_ZEROENTROPY_API_KEY),
|
||||
reranker_zeroentropy_model=os.getenv(ENV_RERANKER_ZEROENTROPY_MODEL, DEFAULT_RERANKER_ZEROENTROPY_MODEL),
|
||||
# Server
|
||||
host=os.getenv(ENV_HOST, DEFAULT_HOST),
|
||||
port=int(os.getenv(ENV_PORT, DEFAULT_PORT)),
|
||||
@@ -983,6 +1051,9 @@ class HindsightConfig:
|
||||
log_level=os.getenv(ENV_LOG_LEVEL, DEFAULT_LOG_LEVEL),
|
||||
log_format=os.getenv(ENV_LOG_FORMAT, DEFAULT_LOG_FORMAT).lower(),
|
||||
mcp_enabled=os.getenv(ENV_MCP_ENABLED, str(DEFAULT_MCP_ENABLED)).lower() == "true",
|
||||
mcp_enabled_tools=[t.strip() for t in os.getenv(ENV_MCP_ENABLED_TOOLS).split(",") if t.strip()]
|
||||
if os.getenv(ENV_MCP_ENABLED_TOOLS)
|
||||
else DEFAULT_MCP_ENABLED_TOOLS,
|
||||
enable_bank_config_api=os.getenv(ENV_ENABLE_BANK_CONFIG_API, str(DEFAULT_ENABLE_BANK_CONFIG_API)).lower()
|
||||
== "true",
|
||||
# Recall
|
||||
@@ -1010,6 +1081,7 @@ class HindsightConfig:
|
||||
retain_extraction_mode=_validate_extraction_mode(
|
||||
os.getenv(ENV_RETAIN_EXTRACTION_MODE, DEFAULT_RETAIN_EXTRACTION_MODE)
|
||||
),
|
||||
retain_mission=os.getenv(ENV_RETAIN_MISSION) or DEFAULT_RETAIN_MISSION,
|
||||
retain_custom_instructions=os.getenv(ENV_RETAIN_CUSTOM_INSTRUCTIONS) or DEFAULT_RETAIN_CUSTOM_INSTRUCTIONS,
|
||||
retain_batch_tokens=int(os.getenv(ENV_RETAIN_BATCH_TOKENS, str(DEFAULT_RETAIN_BATCH_TOKENS))),
|
||||
retain_batch_enabled=os.getenv(ENV_RETAIN_BATCH_ENABLED, str(DEFAULT_RETAIN_BATCH_ENABLED)).lower()
|
||||
@@ -1030,6 +1102,8 @@ class HindsightConfig:
|
||||
file_storage_azure_account_name=os.getenv(ENV_FILE_STORAGE_AZURE_ACCOUNT_NAME) or None,
|
||||
file_storage_azure_account_key=os.getenv(ENV_FILE_STORAGE_AZURE_ACCOUNT_KEY) or None,
|
||||
file_parser=os.getenv(ENV_FILE_PARSER, DEFAULT_FILE_PARSER),
|
||||
file_parser_iris_token=os.getenv(ENV_FILE_PARSER_IRIS_TOKEN) or None,
|
||||
file_parser_iris_org_id=os.getenv(ENV_FILE_PARSER_IRIS_ORG_ID) or None,
|
||||
file_conversion_max_batch_size_mb=int(
|
||||
os.getenv(ENV_FILE_CONVERSION_MAX_BATCH_SIZE_MB, str(DEFAULT_FILE_CONVERSION_MAX_BATCH_SIZE_MB))
|
||||
),
|
||||
@@ -1047,9 +1121,15 @@ class HindsightConfig:
|
||||
consolidation_batch_size=int(
|
||||
os.getenv(ENV_CONSOLIDATION_BATCH_SIZE, str(DEFAULT_CONSOLIDATION_BATCH_SIZE))
|
||||
),
|
||||
consolidation_llm_batch_size=int(
|
||||
os.getenv(ENV_CONSOLIDATION_LLM_BATCH_SIZE, str(DEFAULT_CONSOLIDATION_LLM_BATCH_SIZE))
|
||||
),
|
||||
consolidation_max_tokens=int(
|
||||
os.getenv(ENV_CONSOLIDATION_MAX_TOKENS, str(DEFAULT_CONSOLIDATION_MAX_TOKENS))
|
||||
),
|
||||
observations_mission=os.getenv(ENV_OBSERVATIONS_MISSION) or DEFAULT_OBSERVATIONS_MISSION,
|
||||
entity_labels=None,
|
||||
entities_allow_free_form=True,
|
||||
# Database migrations
|
||||
run_migrations_on_startup=os.getenv(ENV_RUN_MIGRATIONS_ON_STARTUP, "true").lower() == "true",
|
||||
# Database connection pool
|
||||
@@ -1069,6 +1149,20 @@ class HindsightConfig:
|
||||
),
|
||||
# Reflect agent settings
|
||||
reflect_max_iterations=int(os.getenv(ENV_REFLECT_MAX_ITERATIONS, str(DEFAULT_REFLECT_MAX_ITERATIONS))),
|
||||
reflect_max_context_tokens=int(
|
||||
os.getenv(ENV_REFLECT_MAX_CONTEXT_TOKENS, str(DEFAULT_REFLECT_MAX_CONTEXT_TOKENS))
|
||||
),
|
||||
reflect_mission=os.getenv(ENV_REFLECT_MISSION) or None,
|
||||
# Disposition settings (None = fall back to DB value)
|
||||
disposition_skepticism=int(os.getenv(ENV_DISPOSITION_SKEPTICISM))
|
||||
if os.getenv(ENV_DISPOSITION_SKEPTICISM)
|
||||
else DEFAULT_DISPOSITION_SKEPTICISM,
|
||||
disposition_literalism=int(os.getenv(ENV_DISPOSITION_LITERALISM))
|
||||
if os.getenv(ENV_DISPOSITION_LITERALISM)
|
||||
else DEFAULT_DISPOSITION_LITERALISM,
|
||||
disposition_empathy=int(os.getenv(ENV_DISPOSITION_EMPATHY))
|
||||
if os.getenv(ENV_DISPOSITION_EMPATHY)
|
||||
else DEFAULT_DISPOSITION_EMPATHY,
|
||||
# OpenTelemetry tracing configuration
|
||||
otel_traces_enabled=os.getenv(ENV_OTEL_TRACES_ENABLED, str(DEFAULT_OTEL_TRACES_ENABLED)).lower()
|
||||
in ("true", "1", "yes"),
|
||||
|
||||
@@ -16,6 +16,7 @@ from typing import Any
|
||||
import asyncpg
|
||||
|
||||
from hindsight_api.config import HindsightConfig, _get_raw_config, normalize_config_dict
|
||||
from hindsight_api.engine.memory_engine import fq_table
|
||||
from hindsight_api.extensions.tenant import TenantExtension
|
||||
from hindsight_api.models import RequestContext
|
||||
|
||||
@@ -149,8 +150,8 @@ class ConfigResolver:
|
||||
try:
|
||||
async with self.pool.acquire() as conn:
|
||||
row = await conn.fetchrow(
|
||||
"""
|
||||
SELECT config FROM banks WHERE bank_id = $1
|
||||
f"""
|
||||
SELECT config FROM {fq_table("banks")} WHERE bank_id = $1
|
||||
""",
|
||||
bank_id,
|
||||
)
|
||||
@@ -241,8 +242,8 @@ class ConfigResolver:
|
||||
# Merge with existing config (JSONB || operator)
|
||||
async with self.pool.acquire() as conn:
|
||||
await conn.execute(
|
||||
"""
|
||||
UPDATE banks
|
||||
f"""
|
||||
UPDATE {fq_table("banks")}
|
||||
SET config = config || $1::jsonb,
|
||||
updated_at = now()
|
||||
WHERE bank_id = $2
|
||||
@@ -262,9 +263,9 @@ class ConfigResolver:
|
||||
"""
|
||||
async with self.pool.acquire() as conn:
|
||||
await conn.execute(
|
||||
"""
|
||||
UPDATE banks
|
||||
SET config = '{}'::jsonb,
|
||||
f"""
|
||||
UPDATE {fq_table("banks")}
|
||||
SET config = '{{}}'::jsonb,
|
||||
updated_at = now()
|
||||
WHERE bank_id = $1
|
||||
""",
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,85 +1,66 @@
|
||||
"""Prompts for the consolidation engine."""
|
||||
|
||||
CONSOLIDATION_SYSTEM_PROMPT = """You are a memory consolidation system. Your job is to convert facts into durable knowledge (observations) and merge with existing knowledge when appropriate.
|
||||
# Default mission when no bank-specific mission is set
|
||||
_DEFAULT_MISSION = "Track every detail: names, numbers, dates, places, and relationships. Prefer specifics over abstractions, never generalise."
|
||||
|
||||
You must output ONLY valid JSON with no markdown code blocks or additional text. However, the "text" field within each observation should use markdown formatting (headers, lists, bold, etc.) for clarity and readability.
|
||||
# Processing rules — always present regardless of mission
|
||||
_PROCESSING_RULES = """Processing rules (always apply):
|
||||
- REDUNDANT: same info worded differently → UPDATE the existing observation.
|
||||
- CONTRADICTION/UPDATE: capture both states with temporal markers ("used to X, now Y").
|
||||
- RESOLVE REFERENCES: when a new fact provides a concrete value resolving a vague placeholder in an existing observation (e.g. "home country", "hometown", "birthplace", "native language", "her ex", "that city"), UPDATE the observation to embed the resolved value explicitly. Example: new fact says "grandma in Sweden" + existing observation says "moved from her home country" → update to "home country is Sweden".
|
||||
- NEVER merge observations about different people or unrelated topics."""
|
||||
|
||||
## EXTRACT DURABLE KNOWLEDGE, NOT EPHEMERAL STATE
|
||||
Facts often describe events or actions. Extract the DURABLE KNOWLEDGE implied by the fact, not the transient state.
|
||||
# Data section — format placeholders {facts_text} and {observations_text} are substituted at call time
|
||||
_BATCH_DATA_SECTION = """
|
||||
NEW FACTS:
|
||||
{facts_text}
|
||||
|
||||
Examples of extracting durable knowledge:
|
||||
- "User moved to Room 203" -> "Room 203 exists" (location exists, not where user is now)
|
||||
- "User visited Acme Corp at Room 105" -> "Acme Corp is located in Room 105"
|
||||
- "User took the elevator to floor 3" -> "Floor 3 is accessible by elevator"
|
||||
- "User met Sarah at the lobby" -> "Sarah can be found at the lobby"
|
||||
|
||||
DO NOT track current user position/state as knowledge - that changes constantly.
|
||||
DO track permanent facts learned from the user's actions.
|
||||
|
||||
## PRESERVE SPECIFIC DETAILS
|
||||
Keep names, locations, numbers, and other specifics. Do NOT:
|
||||
- Abstract into general principles
|
||||
- Generate business insights
|
||||
- Make knowledge generic
|
||||
|
||||
GOOD examples:
|
||||
- Fact: "John likes pizza" -> "John likes pizza"
|
||||
- Fact: "Alice works at Google" -> "Alice works at Google"
|
||||
|
||||
BAD examples:
|
||||
- "John likes pizza" -> "Understanding dietary preferences helps..." (TOO ABSTRACT)
|
||||
- "User is at Room 203" -> "User is currently at Room 203" (EPHEMERAL STATE)
|
||||
|
||||
## MERGE RULES (when comparing to existing observations):
|
||||
1. REDUNDANT: Same information worded differently → update existing
|
||||
2. CONTRADICTION: Opposite information about same topic → update with temporal markers showing change
|
||||
Example: "Alex used to love pizza but now hates it" OR "Alex's pizza preference changed from love to hate"
|
||||
3. UPDATE: New state replacing old state → update showing the transition with "used to", "now", "changed from X to Y"
|
||||
|
||||
## CRITICAL RULES:
|
||||
- NEVER merge facts about DIFFERENT people
|
||||
- NEVER merge unrelated topics (food preferences vs work vs hobbies)
|
||||
- When merging contradictions, the "text" field MUST capture BOTH states with temporal markers:
|
||||
* Use "used to X, now Y" OR "changed from X to Y" OR "X but now Y"
|
||||
* DO NOT just state the new fact - you MUST show the change
|
||||
- Keep observations focused on ONE specific topic per person
|
||||
- The "text" field MUST contain durable knowledge, not ephemeral state
|
||||
- Do NOT include "tags" in output - tags are handled automatically"""
|
||||
|
||||
CONSOLIDATION_USER_PROMPT = """Analyze this new fact and consolidate into knowledge.
|
||||
{mission_section}
|
||||
NEW FACT: {fact_text}
|
||||
|
||||
EXISTING OBSERVATIONS (JSON array with source memories and dates):
|
||||
EXISTING OBSERVATIONS (JSON array, pooled from recalls across all facts above):
|
||||
{observations_text}
|
||||
|
||||
Each observation includes:
|
||||
- id: unique identifier for updating
|
||||
- text: the observation content
|
||||
- proof_count: number of supporting memories
|
||||
- tags: visibility scope (handled automatically)
|
||||
- created_at/updated_at: when observation was created/modified
|
||||
- occurred_start/occurred_end: temporal range of source facts
|
||||
- source_memories: array of supporting facts with their text and dates
|
||||
|
||||
Instructions:
|
||||
1. Extract DURABLE KNOWLEDGE from the new fact (not ephemeral state)
|
||||
2. Review source_memories in existing observations to understand evidence
|
||||
3. Check dates to detect contradictions or updates
|
||||
4. Compare with observations:
|
||||
- Same topic → UPDATE with learning_id
|
||||
- New topic → CREATE new observation
|
||||
- Purely ephemeral → return []
|
||||
Compare the facts against existing observations:
|
||||
- Same topic as an existing observation → UPDATE it (observation_id + source_fact_ids)
|
||||
- New topic with durable knowledge → CREATE a new observation (source_fact_ids)
|
||||
- Cross-reference facts within the batch: a later fact may resolve a vague reference in an earlier one
|
||||
- Purely ephemeral facts → omit them (no create/update needed)"""
|
||||
|
||||
Output JSON array of actions (the "text" field should use markdown formatting for structure):
|
||||
[
|
||||
{{"action": "update", "learning_id": "uuid-from-observations", "text": "## Updated Knowledge\n\n**Key point**: details here\n\n- Supporting detail 1\n- Supporting detail 2", "reason": "..."}},
|
||||
{{"action": "create", "text": "## New Durable Knowledge\n\nDescription with **emphasis** and proper structure", "reason": "..."}}
|
||||
]
|
||||
# Output format — JSON braces escaped as {{ }} so .format() leaves them literal
|
||||
_BATCH_OUTPUT_FORMAT = """
|
||||
Output a JSON object with three arrays.
|
||||
|
||||
Return [] if fact contains no durable knowledge.
|
||||
Example (showing the required UUID format for all IDs):
|
||||
{{"creates": [{{"text": "Alice lives in Berlin", "source_fact_ids": ["a1b2c3d4-e5f6-7890-abcd-ef1234567890", "b2c3d4e5-f6a7-8901-bcde-f12345678901"]}}],
|
||||
"updates": [{{"text": "Alice works at Acme Corp as a senior engineer", "observation_id": "c3d4e5f6-a7b8-9012-cdef-123456789012", "source_fact_ids": ["d4e5f6a7-b8c9-0123-defa-234567890123"]}}],
|
||||
"deletes": [{{"observation_id": "e5f6a7b8-c9d0-1234-efab-345678901234"}}]}}
|
||||
|
||||
IMPORTANT: Format the "text" field with markdown for better readability:
|
||||
- Use headers, lists, bold/italic, tables where appropriate
|
||||
- CRITICAL: Add blank lines before and after block elements (tables, code blocks, lists)
|
||||
- Ensure proper spacing for markdown to render correctly"""
|
||||
Rules:
|
||||
- "source_fact_ids": copy the EXACT UUID strings shown in brackets [uuid] from NEW FACTS — never use integers or positions.
|
||||
- "observation_id": copy the EXACT "id" UUID string from EXISTING OBSERVATIONS.
|
||||
- One create/update may reference multiple facts when they jointly support the observation.
|
||||
- "deletes": only when an observation is directly superseded or contradicted by new facts.
|
||||
- Do NOT include "tags" — handled automatically.
|
||||
- Return {{"creates": [], "updates": [], "deletes": []}} if nothing durable is found."""
|
||||
|
||||
|
||||
def build_batch_consolidation_prompt(observations_mission: str | None = None) -> str:
|
||||
"""
|
||||
Build the consolidation prompt for batch mode (multiple facts per LLM call).
|
||||
|
||||
The mission defines *what* to track (customisable per bank).
|
||||
Processing rules and output format are always present regardless of mission.
|
||||
"""
|
||||
mission = observations_mission or _DEFAULT_MISSION
|
||||
|
||||
return (
|
||||
"You are a memory consolidation system. Synthesize facts into observations "
|
||||
"and merge with existing observations when appropriate.\n\n"
|
||||
f"## MISSION\n{mission}\n\n"
|
||||
f"{_PROCESSING_RULES}" + _BATCH_DATA_SECTION + _BATCH_OUTPUT_FORMAT
|
||||
)
|
||||
|
||||
@@ -29,6 +29,7 @@ from ..config import (
|
||||
DEFAULT_RERANKER_PROVIDER,
|
||||
DEFAULT_RERANKER_TEI_BATCH_SIZE,
|
||||
DEFAULT_RERANKER_TEI_MAX_CONCURRENT,
|
||||
DEFAULT_RERANKER_ZEROENTROPY_MODEL,
|
||||
ENV_RERANKER_COHERE_API_KEY,
|
||||
ENV_RERANKER_COHERE_MODEL,
|
||||
ENV_RERANKER_FLASHRANK_CACHE_DIR,
|
||||
@@ -42,6 +43,7 @@ from ..config import (
|
||||
ENV_RERANKER_TEI_BATCH_SIZE,
|
||||
ENV_RERANKER_TEI_MAX_CONCURRENT,
|
||||
ENV_RERANKER_TEI_URL,
|
||||
ENV_RERANKER_ZEROENTROPY_API_KEY,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -556,6 +558,104 @@ class CohereCrossEncoder(CrossEncoderModel):
|
||||
return all_scores
|
||||
|
||||
|
||||
class ZeroEntropyCrossEncoder(CrossEncoderModel):
|
||||
"""
|
||||
ZeroEntropy cross-encoder implementation using the ZeroEntropy Rerank API.
|
||||
|
||||
Supports zerank-2 (flagship) and zerank-2-small models.
|
||||
See: https://docs.zeroentropy.dev/models
|
||||
"""
|
||||
|
||||
RERANK_URL = "https://api.zeroentropy.dev/v1/models/rerank"
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
api_key: str,
|
||||
model: str = DEFAULT_RERANKER_ZEROENTROPY_MODEL,
|
||||
timeout: float = 60.0,
|
||||
):
|
||||
"""
|
||||
Initialize ZeroEntropy cross-encoder client.
|
||||
|
||||
Args:
|
||||
api_key: ZeroEntropy API key
|
||||
model: ZeroEntropy rerank model name (default: zerank-2)
|
||||
timeout: Request timeout in seconds (default: 60.0)
|
||||
"""
|
||||
self.api_key = api_key
|
||||
self.model = model
|
||||
self.timeout = timeout
|
||||
self._async_client: httpx.AsyncClient | None = None
|
||||
|
||||
@property
|
||||
def provider_name(self) -> str:
|
||||
return "zeroentropy"
|
||||
|
||||
async def initialize(self) -> None:
|
||||
"""Initialize the async HTTP client."""
|
||||
if self._async_client is not None:
|
||||
return
|
||||
|
||||
logger.info(f"Reranker: initializing ZeroEntropy provider with model {self.model}")
|
||||
self._async_client = httpx.AsyncClient(
|
||||
timeout=self.timeout,
|
||||
headers={
|
||||
"Authorization": f"Bearer {self.api_key}",
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
)
|
||||
logger.info("Reranker: ZeroEntropy provider initialized")
|
||||
|
||||
async def predict(self, pairs: list[tuple[str, str]]) -> list[float]:
|
||||
"""
|
||||
Score query-document pairs using the ZeroEntropy Rerank API.
|
||||
|
||||
Args:
|
||||
pairs: List of (query, document) tuples to score
|
||||
|
||||
Returns:
|
||||
List of relevance scores
|
||||
"""
|
||||
if self._async_client is None:
|
||||
raise RuntimeError("Reranker not initialized. Call initialize() first.")
|
||||
|
||||
if not pairs:
|
||||
return []
|
||||
|
||||
# Group pairs by query for efficient batching
|
||||
query_groups: dict[str, list[tuple[int, str]]] = {}
|
||||
for idx, (query, text) in enumerate(pairs):
|
||||
if query not in query_groups:
|
||||
query_groups[query] = []
|
||||
query_groups[query].append((idx, text))
|
||||
|
||||
all_scores = [0.0] * len(pairs)
|
||||
|
||||
for query, indexed_texts in query_groups.items():
|
||||
texts = [text for _, text in indexed_texts]
|
||||
indices = [idx for idx, _ in indexed_texts]
|
||||
|
||||
response = await self._async_client.post(
|
||||
self.RERANK_URL,
|
||||
json={
|
||||
"model": self.model,
|
||||
"query": query,
|
||||
"documents": texts,
|
||||
"top_n": len(texts),
|
||||
},
|
||||
)
|
||||
response.raise_for_status()
|
||||
result = response.json()
|
||||
|
||||
# Map scores back to original positions
|
||||
for item in result.get("results", []):
|
||||
original_idx = item["index"]
|
||||
score = item["relevance_score"]
|
||||
all_scores[indices[original_idx]] = score
|
||||
|
||||
return all_scores
|
||||
|
||||
|
||||
class RRFPassthroughCrossEncoder(CrossEncoderModel):
|
||||
"""
|
||||
Passthrough cross-encoder that preserves RRF scores without neural reranking.
|
||||
@@ -1010,9 +1110,19 @@ def create_cross_encoder_from_env() -> CrossEncoderModel:
|
||||
model=config.reranker_litellm_sdk_model,
|
||||
api_base=config.reranker_litellm_sdk_api_base,
|
||||
)
|
||||
elif provider == "zeroentropy":
|
||||
api_key = config.reranker_zeroentropy_api_key
|
||||
if not api_key:
|
||||
raise ValueError(
|
||||
f"{ENV_RERANKER_ZEROENTROPY_API_KEY} is required when {ENV_RERANKER_PROVIDER} is 'zeroentropy'"
|
||||
)
|
||||
return ZeroEntropyCrossEncoder(
|
||||
api_key=api_key,
|
||||
model=config.reranker_zeroentropy_model,
|
||||
)
|
||||
elif provider == "rrf":
|
||||
return RRFPassthroughCrossEncoder()
|
||||
else:
|
||||
raise ValueError(
|
||||
f"Unknown reranker provider: {provider}. Supported: 'local', 'tei', 'cohere', 'flashrank', 'litellm', 'litellm-sdk', 'rrf'"
|
||||
f"Unknown reranker provider: {provider}. Supported: 'local', 'tei', 'cohere', 'zeroentropy', 'flashrank', 'litellm', 'litellm-sdk', 'rrf'"
|
||||
)
|
||||
|
||||
@@ -794,6 +794,7 @@ class LiteLLMSDKEmbeddings(Embeddings):
|
||||
"model": self.model,
|
||||
"input": ["test"],
|
||||
"api_key": self.api_key,
|
||||
"encoding_format": "float",
|
||||
}
|
||||
if self.api_base:
|
||||
embed_kwargs["api_base"] = self.api_base
|
||||
@@ -840,6 +841,7 @@ class LiteLLMSDKEmbeddings(Embeddings):
|
||||
"model": self.model,
|
||||
"input": batch,
|
||||
"api_key": self.api_key,
|
||||
"encoding_format": "float",
|
||||
}
|
||||
if self.api_base:
|
||||
embed_kwargs["api_base"] = self.api_base
|
||||
|
||||
@@ -12,6 +12,7 @@ import asyncpg
|
||||
|
||||
from .db_utils import acquire_with_retry
|
||||
from .memory_engine import fq_table
|
||||
from .retain.entity_labels import build_labels_lookup as _build_labels_lookup_from_config
|
||||
|
||||
# Load spaCy model (singleton)
|
||||
_nlp = None
|
||||
@@ -31,6 +32,11 @@ class EntityResolver:
|
||||
"""
|
||||
self.pool = pool
|
||||
|
||||
@staticmethod
|
||||
def _build_labels_lookup(entity_labels: list | None) -> set[str]:
|
||||
"""Build a set of valid 'key:value' entity label strings for fast lookup."""
|
||||
return _build_labels_lookup_from_config(entity_labels)
|
||||
|
||||
async def resolve_entities_batch(
|
||||
self,
|
||||
bank_id: str,
|
||||
@@ -38,6 +44,7 @@ class EntityResolver:
|
||||
context: str,
|
||||
unit_event_date,
|
||||
conn=None,
|
||||
entity_labels: list | None = None,
|
||||
) -> list[str]:
|
||||
"""
|
||||
Resolve multiple entities in batch (MUCH faster than sequential).
|
||||
@@ -58,14 +65,25 @@ class EntityResolver:
|
||||
if not entities_data:
|
||||
return []
|
||||
|
||||
taxonomy_lookup = self._build_labels_lookup(entity_labels)
|
||||
if conn is None:
|
||||
async with acquire_with_retry(self.pool) as conn:
|
||||
return await self._resolve_entities_batch_impl(conn, bank_id, entities_data, context, unit_event_date)
|
||||
return await self._resolve_entities_batch_impl(
|
||||
conn, bank_id, entities_data, context, unit_event_date, taxonomy_lookup
|
||||
)
|
||||
else:
|
||||
return await self._resolve_entities_batch_impl(conn, bank_id, entities_data, context, unit_event_date)
|
||||
return await self._resolve_entities_batch_impl(
|
||||
conn, bank_id, entities_data, context, unit_event_date, taxonomy_lookup
|
||||
)
|
||||
|
||||
async def _resolve_entities_batch_impl(
|
||||
self, conn, bank_id: str, entities_data: list[dict], context: str, unit_event_date
|
||||
self,
|
||||
conn,
|
||||
bank_id: str,
|
||||
entities_data: list[dict],
|
||||
context: str,
|
||||
unit_event_date,
|
||||
taxonomy_lookup: set[str] | None = None,
|
||||
) -> list[str]:
|
||||
# Query ALL candidates for this bank
|
||||
all_entities = await conn.fetch(
|
||||
@@ -135,12 +153,19 @@ class EntityResolver:
|
||||
entities_to_update = [] # (entity_id, event_date)
|
||||
entities_to_create = [] # (idx, entity_data, event_date)
|
||||
|
||||
taxonomy_lookup = taxonomy_lookup or set()
|
||||
|
||||
for idx, entity_data in enumerate(entities_data):
|
||||
entity_text = entity_data["text"]
|
||||
nearby_entities = entity_data.get("nearby_entities", [])
|
||||
# Use per-entity date if available, otherwise fall back to batch-level date
|
||||
entity_event_date = entity_data.get("event_date", unit_event_date)
|
||||
|
||||
# Taxonomy entities: skip fuzzy matching, use exact canonical name
|
||||
if taxonomy_lookup and entity_text.lower() in taxonomy_lookup:
|
||||
entities_to_create.append((idx, entity_data, entity_event_date))
|
||||
continue
|
||||
|
||||
candidates = all_candidates.get(entity_text, [])
|
||||
|
||||
if not candidates:
|
||||
@@ -237,7 +262,7 @@ class EntityResolver:
|
||||
rows = await conn.fetch(
|
||||
f"""
|
||||
INSERT INTO {fq_table("entities")} (bank_id, canonical_name, first_seen, last_seen, mention_count)
|
||||
SELECT $1, name, event_date, event_date, cnt
|
||||
SELECT $1, name, COALESCE(event_date, now()), COALESCE(event_date, now()), cnt
|
||||
FROM unnest($2::text[], $3::timestamptz[], $4::int[]) AS t(name, event_date, cnt)
|
||||
ON CONFLICT (bank_id, LOWER(canonical_name))
|
||||
DO UPDATE SET
|
||||
@@ -408,7 +433,7 @@ class EntityResolver:
|
||||
entity_id = await conn.fetchval(
|
||||
f"""
|
||||
INSERT INTO {fq_table("entities")} (bank_id, canonical_name, first_seen, last_seen, mention_count)
|
||||
VALUES ($1, $2, $3, $4, 1)
|
||||
VALUES ($1, $2, COALESCE($3, now()), COALESCE($4, now()), 1)
|
||||
ON CONFLICT (bank_id, LOWER(canonical_name))
|
||||
DO UPDATE SET
|
||||
mention_count = {fq_table("entities")}.mention_count + 1,
|
||||
|
||||
@@ -60,6 +60,59 @@ class OutputTooLongError(Exception):
|
||||
pass
|
||||
|
||||
|
||||
def parse_llm_json(raw: str) -> Any:
|
||||
"""
|
||||
Robustly parse JSON returned by an LLM.
|
||||
|
||||
Handles common LLM output quirks:
|
||||
1. Markdown code fences (```json ... ```) — strip them before parsing.
|
||||
2. Embedded control characters (\\x00-\\x1f, \\x7f) — replace with space
|
||||
and retry if the initial parse fails.
|
||||
|
||||
Args:
|
||||
raw: Raw text returned by the LLM.
|
||||
|
||||
Returns:
|
||||
Parsed Python object (dict, list, etc.).
|
||||
|
||||
Raises:
|
||||
json.JSONDecodeError: If the text cannot be parsed even after cleanup.
|
||||
"""
|
||||
text = raw.strip()
|
||||
|
||||
# Strip markdown code fences (some models wrap JSON in ```json ... ```)
|
||||
if text.startswith("```"):
|
||||
text = text.split("\n", 1)[1] if "\n" in text else text[3:]
|
||||
if text.endswith("```"):
|
||||
text = text[:-3]
|
||||
text = text.strip()
|
||||
|
||||
try:
|
||||
return json.loads(text)
|
||||
except json.JSONDecodeError:
|
||||
# Some models (e.g. Gemini) embed raw control characters inside JSON
|
||||
# string values. Replacing them with a space usually produces valid JSON.
|
||||
cleaned = re.sub(r"[\x00-\x1f\x7f]", " ", text)
|
||||
return json.loads(cleaned)
|
||||
|
||||
|
||||
_PROVIDERS_WITHOUT_API_KEY = frozenset(
|
||||
{
|
||||
"ollama",
|
||||
"lmstudio",
|
||||
"openai-codex",
|
||||
"claude-code",
|
||||
"mock",
|
||||
"vertexai",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def requires_api_key(provider: str) -> bool:
|
||||
"""Return True if the given provider requires an API key to operate."""
|
||||
return provider.lower() not in _PROVIDERS_WITHOUT_API_KEY
|
||||
|
||||
|
||||
def create_llm_provider(
|
||||
provider: str,
|
||||
api_key: str,
|
||||
@@ -552,8 +605,9 @@ class LLMProvider:
|
||||
provider = os.getenv("HINDSIGHT_API_LLM_PROVIDER", "groq")
|
||||
api_key = os.getenv("HINDSIGHT_API_LLM_API_KEY", "")
|
||||
|
||||
# API key not needed for openai-codex (uses OAuth) or claude-code (uses Keychain OAuth)
|
||||
if not api_key and provider not in ("openai-codex", "claude-code"):
|
||||
# API key not needed for openai-codex (uses OAuth), claude-code (uses Keychain OAuth),
|
||||
# ollama (local), or vertexai (uses GCP service account credentials)
|
||||
if not api_key and provider not in ("openai-codex", "claude-code", "ollama", "vertexai"):
|
||||
raise ValueError(
|
||||
"HINDSIGHT_API_LLM_API_KEY environment variable is required (unless using openai-codex or claude-code)"
|
||||
)
|
||||
@@ -569,8 +623,9 @@ class LLMProvider:
|
||||
provider = os.getenv("HINDSIGHT_API_ANSWER_LLM_PROVIDER", os.getenv("HINDSIGHT_API_LLM_PROVIDER", "groq"))
|
||||
api_key = os.getenv("HINDSIGHT_API_ANSWER_LLM_API_KEY", os.getenv("HINDSIGHT_API_LLM_API_KEY", ""))
|
||||
|
||||
# API key not needed for openai-codex (uses OAuth) or claude-code (uses Keychain OAuth)
|
||||
if not api_key and provider not in ("openai-codex", "claude-code"):
|
||||
# API key not needed for openai-codex (uses OAuth), claude-code (uses Keychain OAuth),
|
||||
# ollama (local), or vertexai (uses GCP service account credentials)
|
||||
if not api_key and provider not in ("openai-codex", "claude-code", "ollama", "vertexai"):
|
||||
raise ValueError(
|
||||
"HINDSIGHT_API_LLM_API_KEY or HINDSIGHT_API_ANSWER_LLM_API_KEY environment variable is required "
|
||||
"(unless using openai-codex or claude-code)"
|
||||
@@ -587,8 +642,9 @@ class LLMProvider:
|
||||
provider = os.getenv("HINDSIGHT_API_JUDGE_LLM_PROVIDER", os.getenv("HINDSIGHT_API_LLM_PROVIDER", "groq"))
|
||||
api_key = os.getenv("HINDSIGHT_API_JUDGE_LLM_API_KEY", os.getenv("HINDSIGHT_API_LLM_API_KEY", ""))
|
||||
|
||||
# API key not needed for openai-codex (uses OAuth) or claude-code (uses Keychain OAuth)
|
||||
if not api_key and provider not in ("openai-codex", "claude-code"):
|
||||
# API key not needed for openai-codex (uses OAuth), claude-code (uses Keychain OAuth),
|
||||
# ollama (local), or vertexai (uses GCP service account credentials)
|
||||
if not api_key and provider not in ("openai-codex", "claude-code", "ollama", "vertexai"):
|
||||
raise ValueError(
|
||||
"HINDSIGHT_API_LLM_API_KEY or HINDSIGHT_API_JUDGE_LLM_API_KEY environment variable is required "
|
||||
"(unless using openai-codex or claude-code)"
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,9 +1,10 @@
|
||||
"""File parser implementations."""
|
||||
|
||||
from .base import FileParser
|
||||
from .base import FileParser, UnsupportedFileTypeError
|
||||
from .iris import IrisParser
|
||||
from .markitdown import MarkitdownParser
|
||||
|
||||
__all__ = ["FileParser", "MarkitdownParser", "FileParserRegistry"]
|
||||
__all__ = ["FileParser", "UnsupportedFileTypeError", "IrisParser", "MarkitdownParser", "FileParserRegistry"]
|
||||
|
||||
|
||||
class FileParserRegistry:
|
||||
@@ -43,7 +44,8 @@ class FileParserRegistry:
|
||||
ValueError: If no suitable parser found
|
||||
"""
|
||||
if name:
|
||||
# Explicit parser requested
|
||||
# Explicit parser requested — return it directly, let the parser
|
||||
# raise UnsupportedFileTypeError from convert() if needed
|
||||
if name not in self._parsers:
|
||||
raise ValueError(f"Parser '{name}' not found. Available: {list(self._parsers.keys())}")
|
||||
return self._parsers[name]
|
||||
|
||||
@@ -3,6 +3,12 @@
|
||||
from abc import ABC, abstractmethod
|
||||
|
||||
|
||||
class UnsupportedFileTypeError(Exception):
|
||||
"""Raised by a parser when it does not support the given file type."""
|
||||
|
||||
pass
|
||||
|
||||
|
||||
class FileParser(ABC):
|
||||
"""Abstract base for file to markdown parsers."""
|
||||
|
||||
@@ -19,24 +25,27 @@ class FileParser(ABC):
|
||||
Markdown content as string
|
||||
|
||||
Raises:
|
||||
ValueError: If file format is not supported
|
||||
RuntimeError: If parsing fails
|
||||
UnsupportedFileTypeError: If the file type is not supported by this parser
|
||||
RuntimeError: If parsing fails for another reason
|
||||
"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def supports(self, filename: str, content_type: str | None = None) -> bool:
|
||||
"""
|
||||
Check if parser supports this file type.
|
||||
|
||||
Override this for local/static extension-based filtering.
|
||||
Parsers that delegate to a remote service should leave this as True
|
||||
and raise UnsupportedFileTypeError from convert() instead.
|
||||
|
||||
Args:
|
||||
filename: File name (used for extension check)
|
||||
content_type: MIME type (optional)
|
||||
|
||||
Returns:
|
||||
True if this parser can handle the file
|
||||
True if this parser can handle the file (default: True)
|
||||
"""
|
||||
pass
|
||||
return True
|
||||
|
||||
@abstractmethod
|
||||
def name(self) -> str:
|
||||
|
||||
@@ -0,0 +1,137 @@
|
||||
"""Iris parser implementation using the Vectorize Iris HTTP API."""
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import mimetypes
|
||||
import time
|
||||
|
||||
import httpx
|
||||
|
||||
from .base import FileParser, UnsupportedFileTypeError
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_IRIS_BASE_URL = "https://api.vectorize.io/v1"
|
||||
_DEFAULT_POLL_INTERVAL = 2.0 # seconds
|
||||
_DEFAULT_TIMEOUT = 300.0 # seconds
|
||||
|
||||
|
||||
class IrisParser(FileParser):
|
||||
"""
|
||||
Iris file parser using the Vectorize Iris cloud extraction service.
|
||||
|
||||
Uploads files to the Vectorize Iris API, starts an extraction job,
|
||||
and polls until the text is ready. The API determines which file types
|
||||
are supported — UnsupportedFileTypeError is raised if the file is rejected.
|
||||
|
||||
Authentication:
|
||||
Requires HINDSIGHT_API_FILE_PARSER_IRIS_TOKEN and
|
||||
HINDSIGHT_API_FILE_PARSER_IRIS_ORG_ID environment variables,
|
||||
or pass them explicitly via the constructor.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
token: str,
|
||||
org_id: str,
|
||||
poll_interval: float = _DEFAULT_POLL_INTERVAL,
|
||||
timeout: float = _DEFAULT_TIMEOUT,
|
||||
):
|
||||
"""
|
||||
Initialize iris parser.
|
||||
|
||||
Args:
|
||||
token: Vectorize API token
|
||||
org_id: Vectorize organization ID
|
||||
poll_interval: Seconds between status poll requests (default: 2)
|
||||
timeout: Maximum seconds to wait for extraction (default: 300)
|
||||
"""
|
||||
self._token = token
|
||||
self._org_id = org_id
|
||||
self._poll_interval = poll_interval
|
||||
self._timeout = timeout
|
||||
self._auth_headers = {"Authorization": f"Bearer {token}"}
|
||||
|
||||
async def convert(self, file_data: bytes, filename: str) -> str:
|
||||
"""
|
||||
Parse file to text using the Vectorize Iris API.
|
||||
|
||||
Raises:
|
||||
UnsupportedFileTypeError: If the Iris API rejects the file type (4xx)
|
||||
RuntimeError: If extraction fails for another reason
|
||||
"""
|
||||
content_type = mimetypes.guess_type(filename)[0] or "application/octet-stream"
|
||||
|
||||
async with httpx.AsyncClient() as client:
|
||||
# Step 1: Request a presigned upload URL
|
||||
init_resp = await client.post(
|
||||
f"{_IRIS_BASE_URL}/org/{self._org_id}/files",
|
||||
headers=self._auth_headers,
|
||||
json={"name": filename, "contentType": content_type},
|
||||
)
|
||||
_raise_for_status(init_resp, filename, "file upload init")
|
||||
init_data = init_resp.json()
|
||||
file_id: str = init_data["fileId"]
|
||||
upload_url: str = init_data["uploadUrl"]
|
||||
|
||||
# Step 2: Upload the file bytes to the presigned URL (no auth header)
|
||||
upload_resp = await client.put(
|
||||
upload_url,
|
||||
content=file_data,
|
||||
headers={"Content-Type": content_type},
|
||||
)
|
||||
_raise_for_status(upload_resp, filename, "file upload")
|
||||
|
||||
# Step 3: Start extraction
|
||||
extract_resp = await client.post(
|
||||
f"{_IRIS_BASE_URL}/org/{self._org_id}/extraction",
|
||||
headers=self._auth_headers,
|
||||
json={"fileId": file_id},
|
||||
)
|
||||
_raise_for_status(extract_resp, filename, "start extraction")
|
||||
extraction_id: str = extract_resp.json()["extractionId"]
|
||||
|
||||
# Step 4: Poll until ready or timeout
|
||||
deadline = time.monotonic() + self._timeout
|
||||
while True:
|
||||
status_resp = await client.get(
|
||||
f"{_IRIS_BASE_URL}/org/{self._org_id}/extraction/{extraction_id}",
|
||||
headers=self._auth_headers,
|
||||
)
|
||||
_raise_for_status(status_resp, filename, "poll extraction status")
|
||||
status_data = status_resp.json()
|
||||
|
||||
if status_data.get("ready"):
|
||||
data = status_data.get("data", {})
|
||||
if not data.get("success"):
|
||||
error = data.get("error", "unknown error")
|
||||
raise RuntimeError(f"Iris extraction failed for '{filename}': {error}")
|
||||
text = data.get("text")
|
||||
if not text:
|
||||
raise RuntimeError(f"No content extracted from '{filename}'")
|
||||
return text
|
||||
|
||||
if time.monotonic() >= deadline:
|
||||
raise RuntimeError(f"Iris extraction timed out after {self._timeout}s for '{filename}'")
|
||||
|
||||
await asyncio.sleep(self._poll_interval)
|
||||
|
||||
def name(self) -> str:
|
||||
"""Get parser name."""
|
||||
return "iris"
|
||||
|
||||
|
||||
def _raise_for_status(response: httpx.Response, filename: str, step: str) -> None:
|
||||
"""
|
||||
Raise an appropriate error including the response body on HTTP errors.
|
||||
|
||||
Raises UnsupportedFileTypeError for 4xx responses (file rejected by the API),
|
||||
RuntimeError for other HTTP errors.
|
||||
"""
|
||||
if not response.is_error:
|
||||
return
|
||||
body = response.text or "<empty>"
|
||||
msg = f"Iris API error during {step} for '{filename}': {response.status_code} {response.reason_phrase} — {body}"
|
||||
if response.is_client_error:
|
||||
raise UnsupportedFileTypeError(msg)
|
||||
raise RuntimeError(msg)
|
||||
@@ -238,21 +238,24 @@ class ClaudeCodeLLM(LLMInterface):
|
||||
)
|
||||
|
||||
# Record trace span
|
||||
from hindsight_api.tracing import get_span_recorder
|
||||
try:
|
||||
from hindsight_api.tracing import get_span_recorder
|
||||
|
||||
span_recorder = get_span_recorder()
|
||||
span_recorder.record_llm_call(
|
||||
provider=self.provider,
|
||||
model=self.model,
|
||||
scope=scope,
|
||||
messages=messages,
|
||||
response_content=result if isinstance(result, str) else json.dumps(result),
|
||||
input_tokens=estimated_input,
|
||||
output_tokens=estimated_output,
|
||||
duration=duration,
|
||||
finish_reason=None,
|
||||
error=None,
|
||||
)
|
||||
span_recorder = get_span_recorder()
|
||||
span_recorder.record_llm_call(
|
||||
provider=self.provider,
|
||||
model=self.model,
|
||||
scope=scope,
|
||||
messages=messages,
|
||||
response_content=result if isinstance(result, str) else result.model_dump_json(),
|
||||
input_tokens=estimated_input,
|
||||
output_tokens=estimated_output,
|
||||
duration=duration,
|
||||
finish_reason=None,
|
||||
error=None,
|
||||
)
|
||||
except Exception:
|
||||
pass # logging failure must never affect the operation
|
||||
|
||||
# Log slow calls
|
||||
if duration > 10.0:
|
||||
|
||||
@@ -18,6 +18,7 @@ from google.genai import errors as genai_errors
|
||||
from google.genai import types as genai_types
|
||||
|
||||
from hindsight_api.engine.llm_interface import LLMInterface, OutputTooLongError
|
||||
from hindsight_api.engine.llm_wrapper import parse_llm_json
|
||||
from hindsight_api.engine.response_models import LLMToolCall, LLMToolCallResult, TokenUsage
|
||||
from hindsight_api.metrics import get_metrics_collector
|
||||
|
||||
@@ -221,10 +222,13 @@ class GeminiLLM(LLMInterface):
|
||||
|
||||
for attempt in range(max_retries + 1):
|
||||
try:
|
||||
response = await self._client.aio.models.generate_content(
|
||||
model=self.model,
|
||||
contents=gemini_contents,
|
||||
config=generation_config,
|
||||
response = await asyncio.wait_for(
|
||||
self._client.aio.models.generate_content(
|
||||
model=self.model,
|
||||
contents=gemini_contents,
|
||||
config=generation_config,
|
||||
),
|
||||
timeout=90.0, # Safety net for network hangs; valid slow responses are <90s
|
||||
)
|
||||
|
||||
content = response.text
|
||||
@@ -247,7 +251,7 @@ class GeminiLLM(LLMInterface):
|
||||
|
||||
# Parse structured output if requested
|
||||
if response_format is not None:
|
||||
json_data = json.loads(content)
|
||||
json_data = parse_llm_json(content)
|
||||
if skip_validation:
|
||||
result = json_data
|
||||
else:
|
||||
@@ -405,31 +409,57 @@ class GeminiLLM(LLMInterface):
|
||||
# Convert messages
|
||||
system_instruction = None
|
||||
gemini_contents = []
|
||||
for msg in messages:
|
||||
msg_list = list(messages)
|
||||
i = 0
|
||||
while i < len(msg_list):
|
||||
msg = msg_list[i]
|
||||
role = msg.get("role", "user")
|
||||
content = msg.get("content", "")
|
||||
|
||||
if role == "system":
|
||||
system_instruction = (system_instruction + "\n\n" + content) if system_instruction else content
|
||||
i += 1
|
||||
elif role == "tool":
|
||||
# Gemini uses function_response
|
||||
gemini_contents.append(
|
||||
genai_types.Content(
|
||||
role="user",
|
||||
parts=[
|
||||
genai_types.Part(
|
||||
function_response=genai_types.FunctionResponse(
|
||||
name=msg.get("name", ""),
|
||||
response={"result": content},
|
||||
)
|
||||
# Gemini requires ALL tool responses for a given model turn to be grouped
|
||||
# into a single Content with multiple FunctionResponse parts.
|
||||
# Consecutive role="tool" messages correspond to one model turn's tool calls.
|
||||
parts = []
|
||||
while i < len(msg_list) and msg_list[i].get("role") == "tool":
|
||||
tool_msg = msg_list[i]
|
||||
tool_content = tool_msg.get("content", "")
|
||||
parts.append(
|
||||
genai_types.Part(
|
||||
function_response=genai_types.FunctionResponse(
|
||||
name=tool_msg.get("name", ""),
|
||||
response={"result": tool_content},
|
||||
)
|
||||
],
|
||||
)
|
||||
)
|
||||
)
|
||||
i += 1
|
||||
gemini_contents.append(genai_types.Content(role="user", parts=parts))
|
||||
elif role == "assistant":
|
||||
gemini_contents.append(genai_types.Content(role="model", parts=[genai_types.Part(text=content)]))
|
||||
tool_calls_in_msg = msg.get("tool_calls", [])
|
||||
if tool_calls_in_msg:
|
||||
# Convert OpenAI-style tool_calls to Gemini function_call parts
|
||||
# This is required for proper multi-turn conversation history
|
||||
parts = []
|
||||
if content:
|
||||
parts.append(genai_types.Part(text=content))
|
||||
for tc in tool_calls_in_msg:
|
||||
fn = tc.get("function", {})
|
||||
fn_name = fn.get("name", "")
|
||||
fn_args_str = fn.get("arguments", "{}")
|
||||
fn_args = parse_llm_json(fn_args_str)
|
||||
parts.append(
|
||||
genai_types.Part(function_call=genai_types.FunctionCall(name=fn_name, args=fn_args))
|
||||
)
|
||||
gemini_contents.append(genai_types.Content(role="model", parts=parts))
|
||||
else:
|
||||
gemini_contents.append(genai_types.Content(role="model", parts=[genai_types.Part(text=content)]))
|
||||
i += 1
|
||||
else:
|
||||
gemini_contents.append(genai_types.Content(role="user", parts=[genai_types.Part(text=content)]))
|
||||
i += 1
|
||||
|
||||
config_kwargs: dict[str, Any] = {"tools": gemini_tools}
|
||||
if system_instruction:
|
||||
@@ -437,15 +467,40 @@ class GeminiLLM(LLMInterface):
|
||||
if temperature is not None:
|
||||
config_kwargs["temperature"] = temperature
|
||||
|
||||
# Map OpenAI-style tool_choice to Gemini FunctionCallingConfig
|
||||
if tool_choice == "required":
|
||||
config_kwargs["tool_config"] = genai_types.ToolConfig(
|
||||
function_calling_config=genai_types.FunctionCallingConfig(
|
||||
mode="ANY",
|
||||
)
|
||||
)
|
||||
elif isinstance(tool_choice, dict) and tool_choice.get("type") == "function":
|
||||
fn_name = tool_choice.get("function", {}).get("name")
|
||||
if fn_name:
|
||||
config_kwargs["tool_config"] = genai_types.ToolConfig(
|
||||
function_calling_config=genai_types.FunctionCallingConfig(
|
||||
mode="ANY",
|
||||
allowed_function_names=[fn_name],
|
||||
)
|
||||
)
|
||||
elif tool_choice == "none":
|
||||
config_kwargs["tool_config"] = genai_types.ToolConfig(
|
||||
function_calling_config=genai_types.FunctionCallingConfig(mode="NONE")
|
||||
)
|
||||
# "auto" is the default (no tool_config needed)
|
||||
|
||||
config = genai_types.GenerateContentConfig(**config_kwargs)
|
||||
|
||||
last_exception = None
|
||||
for attempt in range(max_retries + 1):
|
||||
try:
|
||||
response = await self._client.aio.models.generate_content(
|
||||
model=self.model,
|
||||
contents=gemini_contents,
|
||||
config=config,
|
||||
response = await asyncio.wait_for(
|
||||
self._client.aio.models.generate_content(
|
||||
model=self.model,
|
||||
contents=gemini_contents,
|
||||
config=config,
|
||||
),
|
||||
timeout=90.0, # Safety net for network hangs; valid slow responses are <90s
|
||||
)
|
||||
|
||||
# Extract content and tool calls
|
||||
|
||||
@@ -14,32 +14,26 @@ import re
|
||||
import time
|
||||
from typing import TYPE_CHECKING, Any, Awaitable, Callable
|
||||
|
||||
import tiktoken
|
||||
|
||||
from .models import DirectiveInfo, LLMCall, ReflectAgentResult, TokenUsageSummary, ToolCall
|
||||
from .prompts import FINAL_SYSTEM_PROMPT, _extract_directive_rules, build_final_prompt, build_system_prompt_for_tools
|
||||
from .tools_schema import get_reflect_tools
|
||||
|
||||
|
||||
def _build_directives_applied(directives: list[dict[str, Any]] | None) -> list[DirectiveInfo]:
|
||||
"""Build list of DirectiveInfo from directive mental models.
|
||||
|
||||
Handles multiple directive formats:
|
||||
1. New format: directives have direct 'content' field
|
||||
2. Fallback: directives have 'description' field
|
||||
"""
|
||||
"""Build list of DirectiveInfo from directives."""
|
||||
if not directives:
|
||||
return []
|
||||
|
||||
result = []
|
||||
for directive in directives:
|
||||
directive_id = directive.get("id", "")
|
||||
directive_name = directive.get("name", "")
|
||||
|
||||
# Get content from 'content' field or fallback to 'description'
|
||||
content = directive.get("content", "") or directive.get("description", "")
|
||||
|
||||
result.append(DirectiveInfo(id=directive_id, name=directive_name, content=content))
|
||||
|
||||
return result
|
||||
return [
|
||||
DirectiveInfo(
|
||||
id=directive.get("id", ""),
|
||||
name=directive.get("name", ""),
|
||||
content=directive.get("content", ""),
|
||||
)
|
||||
for directive in directives
|
||||
]
|
||||
|
||||
|
||||
if TYPE_CHECKING:
|
||||
@@ -267,6 +261,46 @@ OUTPUT:"""
|
||||
return None, 0, 0
|
||||
|
||||
|
||||
_TIKTOKEN_ENCODING = tiktoken.get_encoding("cl100k_base")
|
||||
|
||||
|
||||
def _count_messages_tokens(messages: list[dict[str, Any]]) -> int:
|
||||
"""Estimate the token count of the messages list using cl100k_base encoding."""
|
||||
total = 0
|
||||
for msg in messages:
|
||||
content = msg.get("content") or ""
|
||||
if isinstance(content, str):
|
||||
total += len(_TIKTOKEN_ENCODING.encode(content))
|
||||
elif isinstance(content, list):
|
||||
for part in content:
|
||||
if isinstance(part, dict) and isinstance(part.get("text"), str):
|
||||
total += len(_TIKTOKEN_ENCODING.encode(part["text"]))
|
||||
# Tool call arguments and results also count
|
||||
for tc in msg.get("tool_calls") or []:
|
||||
if isinstance(tc, dict):
|
||||
func = tc.get("function", {})
|
||||
total += len(_TIKTOKEN_ENCODING.encode(func.get("arguments", "")))
|
||||
return total
|
||||
|
||||
|
||||
def _is_context_overflow_error(exc: Exception) -> bool:
|
||||
"""Return True if the exception signals the LLM context window was exceeded."""
|
||||
msg = str(exc).lower()
|
||||
return any(
|
||||
phrase in msg
|
||||
for phrase in (
|
||||
"context_length_exceeded",
|
||||
"context length exceeded",
|
||||
"maximum context length",
|
||||
"prompt_too_long",
|
||||
"prompt is too long",
|
||||
"resource_exhausted",
|
||||
"input is too long",
|
||||
"too many tokens",
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
async def run_reflect_agent(
|
||||
llm_config: "LLMProvider",
|
||||
bank_id: str,
|
||||
@@ -274,7 +308,7 @@ async def run_reflect_agent(
|
||||
bank_profile: dict[str, Any],
|
||||
search_mental_models_fn: Callable[[str, int], Awaitable[dict[str, Any]]],
|
||||
search_observations_fn: Callable[[str, int], Awaitable[dict[str, Any]]],
|
||||
recall_fn: Callable[[str, int], Awaitable[dict[str, Any]]],
|
||||
recall_fn: Callable[[str, int, int], Awaitable[dict[str, Any]]],
|
||||
expand_fn: Callable[[list[str], str], Awaitable[dict[str, Any]]],
|
||||
context: str | None = None,
|
||||
max_iterations: int = DEFAULT_MAX_ITERATIONS,
|
||||
@@ -283,6 +317,7 @@ async def run_reflect_agent(
|
||||
directives: list[dict[str, Any]] | None = None,
|
||||
has_mental_models: bool = False,
|
||||
budget: str | None = None,
|
||||
max_context_tokens: int = 100_000,
|
||||
) -> ReflectAgentResult:
|
||||
"""
|
||||
Execute the reflect agent loop using native tool calling.
|
||||
@@ -390,12 +425,15 @@ async def run_reflect_agent(
|
||||
f"total={elapsed_ms}ms"
|
||||
)
|
||||
|
||||
consecutive_errors = 0
|
||||
for iteration in range(max_iterations):
|
||||
is_last = iteration == max_iterations - 1
|
||||
|
||||
if is_last:
|
||||
# Force text response on last iteration - no tools
|
||||
prompt = build_final_prompt(query, context_history, bank_profile, context)
|
||||
prompt = build_final_prompt(
|
||||
query, context_history, bank_profile, context, max_context_tokens=max_context_tokens
|
||||
)
|
||||
llm_start = time.time()
|
||||
response, usage = await llm_config.call(
|
||||
messages=[
|
||||
@@ -440,17 +478,91 @@ async def run_reflect_agent(
|
||||
directives_applied=directives_applied,
|
||||
)
|
||||
|
||||
# Proactive context-window guard: if accumulated messages would exceed the
|
||||
# configured token budget, bail out early and synthesize from what we have.
|
||||
estimated_tokens = _count_messages_tokens(messages)
|
||||
if estimated_tokens >= max_context_tokens and (
|
||||
bool(available_memory_ids) or bool(available_mental_model_ids) or bool(available_observation_ids)
|
||||
):
|
||||
logger.warning(
|
||||
f"[REFLECT {reflect_id}] Context budget exceeded on iteration {iteration + 1}: "
|
||||
f"~{estimated_tokens} tokens >= {max_context_tokens} limit. Forcing final synthesis."
|
||||
)
|
||||
prompt = build_final_prompt(
|
||||
query, context_history, bank_profile, context, max_context_tokens=max_context_tokens
|
||||
)
|
||||
llm_start = time.time()
|
||||
response, usage = await llm_config.call(
|
||||
messages=[
|
||||
{"role": "system", "content": FINAL_SYSTEM_PROMPT},
|
||||
{"role": "user", "content": prompt},
|
||||
],
|
||||
scope="reflect",
|
||||
max_completion_tokens=max_tokens,
|
||||
return_usage=True,
|
||||
)
|
||||
llm_duration = int((time.time() - llm_start) * 1000)
|
||||
total_input_tokens += usage.input_tokens
|
||||
total_output_tokens += usage.output_tokens
|
||||
llm_trace.append(
|
||||
{
|
||||
"scope": "final",
|
||||
"duration_ms": llm_duration,
|
||||
"input_tokens": usage.input_tokens,
|
||||
"output_tokens": usage.output_tokens,
|
||||
}
|
||||
)
|
||||
answer = _clean_answer_text(response.strip())
|
||||
|
||||
structured_output = None
|
||||
if response_schema and answer:
|
||||
structured_output, struct_in, struct_out = await _generate_structured_output(
|
||||
answer, response_schema, llm_config, reflect_id
|
||||
)
|
||||
total_input_tokens += struct_in
|
||||
total_output_tokens += struct_out
|
||||
|
||||
_log_completion(answer, iteration + 1, forced=True)
|
||||
return ReflectAgentResult(
|
||||
text=answer,
|
||||
structured_output=structured_output,
|
||||
iterations=iteration + 1,
|
||||
tools_called=total_tools_called,
|
||||
tool_trace=tool_trace,
|
||||
llm_trace=_get_llm_trace(),
|
||||
usage=_get_usage(),
|
||||
directives_applied=directives_applied,
|
||||
)
|
||||
|
||||
# Call LLM with tools
|
||||
llm_start = time.time()
|
||||
|
||||
# Determine tool_choice for this iteration.
|
||||
# Force the full hierarchical retrieval path before allowing auto:
|
||||
# With mental models:
|
||||
# 0 → search_mental_models, 1 → search_observations, 2 → recall, 3+ → auto
|
||||
# Without mental models:
|
||||
# 0 → search_observations, 1 → recall, 2+ → auto
|
||||
if iteration == 0 and has_mental_models:
|
||||
iter_tool_choice: str | dict = {"type": "function", "function": {"name": "search_mental_models"}}
|
||||
elif iteration == 0:
|
||||
iter_tool_choice = {"type": "function", "function": {"name": "search_observations"}}
|
||||
elif iteration == 1 and has_mental_models:
|
||||
iter_tool_choice = {"type": "function", "function": {"name": "search_observations"}}
|
||||
elif iteration == 1 or (iteration == 2 and has_mental_models):
|
||||
iter_tool_choice = {"type": "function", "function": {"name": "recall"}}
|
||||
else:
|
||||
iter_tool_choice = "auto"
|
||||
|
||||
try:
|
||||
result = await llm_config.call_with_tools(
|
||||
messages=messages,
|
||||
tools=tools,
|
||||
scope="reflect_tool_call",
|
||||
tool_choice="required" if iteration == 0 else "auto", # Force tool use on first iteration
|
||||
tool_choice=iter_tool_choice,
|
||||
)
|
||||
llm_duration = int((time.time() - llm_start) * 1000)
|
||||
consecutive_errors = 0
|
||||
total_input_tokens += result.input_tokens
|
||||
total_output_tokens += result.output_tokens
|
||||
llm_trace.append(
|
||||
@@ -464,15 +576,25 @@ async def run_reflect_agent(
|
||||
|
||||
except Exception as e:
|
||||
err_duration = int((time.time() - llm_start) * 1000)
|
||||
consecutive_errors += 1
|
||||
logger.warning(f"[REFLECT {reflect_id}] LLM error on iteration {iteration + 1}: {e} ({err_duration}ms)")
|
||||
llm_trace.append({"scope": f"agent_{iteration + 1}_err", "duration_ms": err_duration})
|
||||
# Guardrail: If no evidence gathered yet, retry
|
||||
has_gathered_evidence = (
|
||||
bool(available_memory_ids) or bool(available_mental_model_ids) or bool(available_observation_ids)
|
||||
)
|
||||
if not has_gathered_evidence and iteration < max_iterations - 1:
|
||||
# Context overflow errors must never be retried — retrying would only make them worse.
|
||||
# Skip straight to final synthesis with whatever evidence we have.
|
||||
if _is_context_overflow_error(e):
|
||||
logger.warning(
|
||||
f"[REFLECT {reflect_id}] Context window exceeded on iteration {iteration + 1}, "
|
||||
"forcing final synthesis from gathered evidence."
|
||||
)
|
||||
# For other errors: retry if no evidence yet (but cap consecutive errors to avoid long hangs)
|
||||
elif not has_gathered_evidence and iteration < max_iterations - 1 and consecutive_errors < 2:
|
||||
continue
|
||||
prompt = build_final_prompt(query, context_history, bank_profile, context)
|
||||
prompt = build_final_prompt(
|
||||
query, context_history, bank_profile, context, max_context_tokens=max_context_tokens
|
||||
)
|
||||
llm_start = time.time()
|
||||
response, usage = await llm_config.call(
|
||||
messages=[
|
||||
@@ -543,7 +665,9 @@ async def run_reflect_agent(
|
||||
directives_applied=directives_applied,
|
||||
)
|
||||
# Empty response, force final
|
||||
prompt = build_final_prompt(query, context_history, bank_profile, context)
|
||||
prompt = build_final_prompt(
|
||||
query, context_history, bank_profile, context, max_context_tokens=max_context_tokens
|
||||
)
|
||||
llm_start = time.time()
|
||||
response, usage = await llm_config.call(
|
||||
messages=[
|
||||
@@ -807,9 +931,9 @@ async def _process_done_tool(
|
||||
answer = "No answer provided."
|
||||
|
||||
# Validate IDs (only include IDs that were actually retrieved)
|
||||
used_memory_ids = [mid for mid in args.get("memory_ids", []) if mid in available_memory_ids]
|
||||
used_mental_model_ids = [mid for mid in args.get("mental_model_ids", []) if mid in available_mental_model_ids]
|
||||
used_observation_ids = [oid for oid in args.get("observation_ids", []) if oid in available_observation_ids]
|
||||
used_memory_ids = [mid for mid in (args.get("memory_ids") or []) if mid in available_memory_ids]
|
||||
used_mental_model_ids = [mid for mid in (args.get("mental_model_ids") or []) if mid in available_mental_model_ids]
|
||||
used_observation_ids = [oid for oid in (args.get("observation_ids") or []) if oid in available_observation_ids]
|
||||
|
||||
# Generate structured output if schema provided
|
||||
structured_output = None
|
||||
@@ -845,7 +969,7 @@ async def _execute_tool_with_timing(
|
||||
tc: "LLMToolCall",
|
||||
search_mental_models_fn: Callable[[str, int], Awaitable[dict[str, Any]]],
|
||||
search_observations_fn: Callable[[str, int], Awaitable[dict[str, Any]]],
|
||||
recall_fn: Callable[[str, int], Awaitable[dict[str, Any]]],
|
||||
recall_fn: Callable[[str, int, int], Awaitable[dict[str, Any]]],
|
||||
expand_fn: Callable[[list[str], str], Awaitable[dict[str, Any]]],
|
||||
) -> tuple[dict[str, Any], int]:
|
||||
"""Execute a tool call and return result with timing."""
|
||||
@@ -917,7 +1041,7 @@ async def _execute_tool(
|
||||
args: dict[str, Any],
|
||||
search_mental_models_fn: Callable[[str, int], Awaitable[dict[str, Any]]],
|
||||
search_observations_fn: Callable[[str, int], Awaitable[dict[str, Any]]],
|
||||
recall_fn: Callable[[str, int], Awaitable[dict[str, Any]]],
|
||||
recall_fn: Callable[[str, int, int], Awaitable[dict[str, Any]]],
|
||||
expand_fn: Callable[[list[str], str], Awaitable[dict[str, Any]]],
|
||||
) -> dict[str, Any]:
|
||||
"""Execute a single tool by name."""
|
||||
@@ -943,7 +1067,8 @@ async def _execute_tool(
|
||||
if not query:
|
||||
return {"error": "recall requires a query parameter"}
|
||||
max_tokens = max(int(args.get("max_tokens") or 2048), 1000) # Default 2048, min 1000
|
||||
return await recall_fn(query, max_tokens)
|
||||
max_chunk_tokens = max(int(args.get("max_chunk_tokens") or 1000), 1000) # Always enabled, min 1000
|
||||
return await recall_fn(query, max_tokens, max_chunk_tokens)
|
||||
|
||||
elif tool_name == "expand":
|
||||
memory_ids = args.get("memory_ids", [])
|
||||
@@ -971,9 +1096,9 @@ def _summarize_input(tool_name: str, args: dict[str, Any]) -> str:
|
||||
elif tool_name == "recall":
|
||||
query = args.get("query", "")
|
||||
query_preview = f"'{query[:30]}...'" if len(query) > 30 else f"'{query}'"
|
||||
# Show actual value used (default 2048, min 1000)
|
||||
max_tokens = max(int(args.get("max_tokens") or 2048), 1000)
|
||||
return f"(query={query_preview}, max_tokens={max_tokens})"
|
||||
max_chunk_tokens = max(int(args.get("max_chunk_tokens") or 1000), 1000)
|
||||
return f"(query={query_preview}, max_tokens={max_tokens}, max_chunk_tokens={max_chunk_tokens})"
|
||||
elif tool_name == "expand":
|
||||
memory_ids = args.get("memory_ids", [])
|
||||
depth = args.get("depth", "chunk")
|
||||
|
||||
@@ -10,59 +10,30 @@ The reflect agent uses hierarchical retrieval:
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
import tiktoken
|
||||
|
||||
_TIKTOKEN_ENCODING = tiktoken.get_encoding("cl100k_base")
|
||||
|
||||
# Fraction of max_context_tokens reserved for tool results in the final synthesis prompt.
|
||||
# The remainder covers the system prompt, question, bank context, and output tokens.
|
||||
_FINAL_PROMPT_CONTEXT_FRACTION = 0.8
|
||||
|
||||
|
||||
def _extract_directive_rules(directives: list[dict[str, Any]]) -> list[str]:
|
||||
"""
|
||||
Extract directive rules as a list of strings.
|
||||
|
||||
Args:
|
||||
directives: List of directives with name and content
|
||||
|
||||
Returns:
|
||||
List of directive rule strings
|
||||
"""
|
||||
"""Extract directive rules as a list of strings."""
|
||||
rules = []
|
||||
for directive in directives:
|
||||
directive_name = directive.get("name", "")
|
||||
# New format: directives have direct content field
|
||||
name = directive.get("name", "")
|
||||
content = directive.get("content", "")
|
||||
if content:
|
||||
if directive_name:
|
||||
rules.append(f"**{directive_name}**: {content}")
|
||||
else:
|
||||
rules.append(content)
|
||||
else:
|
||||
# Legacy format: check for observations
|
||||
observations = directive.get("observations", [])
|
||||
if observations:
|
||||
for obs in observations:
|
||||
# Support both Pydantic Observation objects and dicts
|
||||
if hasattr(obs, "title"):
|
||||
title = obs.title
|
||||
obs_content = obs.content
|
||||
else:
|
||||
title = obs.get("title", "")
|
||||
obs_content = obs.get("content", "")
|
||||
if title and obs_content:
|
||||
rules.append(f"**{title}**: {obs_content}")
|
||||
elif obs_content:
|
||||
rules.append(obs_content)
|
||||
elif directive_name:
|
||||
# Fallback to description
|
||||
desc = directive.get("description", "")
|
||||
if desc:
|
||||
rules.append(f"**{directive_name}**: {desc}")
|
||||
rules.append(f"**{name}**: {content}" if name else content)
|
||||
return rules
|
||||
|
||||
|
||||
def build_directives_section(directives: list[dict[str, Any]]) -> str:
|
||||
"""
|
||||
Build the directives section for the system prompt.
|
||||
"""Build the directives section for the system prompt.
|
||||
|
||||
Directives are hard rules that MUST be followed in all responses.
|
||||
|
||||
Args:
|
||||
directives: List of directive mental models with observations
|
||||
"""
|
||||
if not directives:
|
||||
return ""
|
||||
@@ -169,6 +140,12 @@ def build_system_prompt_for_tools(
|
||||
|
||||
parts.extend(
|
||||
[
|
||||
"## LANGUAGE RULE (default - directives take precedence)",
|
||||
"- By default, detect the language of the user's question and respond in that SAME language.",
|
||||
"- If the question is in Chinese, respond in Chinese. If in Japanese, respond in Japanese.",
|
||||
"- IMPORTANT: The DIRECTIVES section above has HIGHER PRIORITY than this rule.",
|
||||
" If a directive specifies a language (e.g. 'Always respond in French'), follow the directive.",
|
||||
"",
|
||||
"## CRITICAL RULES",
|
||||
"- ONLY use information from tool results - no external knowledge or guessing",
|
||||
"- You SHOULD synthesize, infer, and reason from the retrieved memories",
|
||||
@@ -205,6 +182,7 @@ def build_system_prompt_for_tools(
|
||||
"### 3. RAW FACTS (recall) - Ground Truth",
|
||||
"- Individual memories (world facts and experiences)",
|
||||
"- Use when: no mental models/observations exist, they're stale, or you need specific details",
|
||||
"- MANDATORY: If search_mental_models and search_observations both return 0 results, you MUST call recall() before giving up",
|
||||
"- This is the source of truth that other levels are built from",
|
||||
"",
|
||||
]
|
||||
@@ -222,6 +200,7 @@ def build_system_prompt_for_tools(
|
||||
"### 2. RAW FACTS (recall) - Ground Truth",
|
||||
"- Individual memories (world facts and experiences)",
|
||||
"- Use when: no observations exist, they're stale, or you need specific details",
|
||||
"- MANDATORY: If search_observations returns 0 results or count=0, you MUST call recall() before giving up",
|
||||
"- This is the source of truth that observations are built from",
|
||||
"",
|
||||
]
|
||||
@@ -299,7 +278,7 @@ def build_system_prompt_for_tools(
|
||||
parts.extend(
|
||||
[
|
||||
"1. First, try search_observations() - check for consolidated knowledge",
|
||||
"2. If observations are stale OR you need specific details, use recall() for raw facts",
|
||||
"2. If search_observations returns 0 results OR observations are stale, you MUST call recall() for raw facts",
|
||||
"3. Use expand() if you need more context on specific memories",
|
||||
"4. When ready, call done() with your answer and supporting IDs",
|
||||
]
|
||||
@@ -315,6 +294,7 @@ def build_system_prompt_for_tools(
|
||||
"- Format for clarity and readability with proper spacing and hierarchy",
|
||||
"- NEVER include memory IDs, UUIDs, or 'Memory references' in the answer text",
|
||||
"- Put IDs ONLY in the memory_ids/mental_model_ids/observation_ids arrays, not in the answer",
|
||||
"- CRITICAL: This is a NON-CONVERSATIONAL system. NEVER ask follow-up questions, offer further assistance, or suggest next steps. Your answer must be complete and self-contained. The user cannot reply.",
|
||||
]
|
||||
)
|
||||
|
||||
@@ -422,6 +402,7 @@ def build_final_prompt(
|
||||
context_history: list[dict],
|
||||
bank_profile: dict,
|
||||
additional_context: str | None = None,
|
||||
max_context_tokens: int = 100_000,
|
||||
) -> str:
|
||||
"""Build the final prompt when forcing a text response (no tools)."""
|
||||
parts = []
|
||||
@@ -451,18 +432,32 @@ def build_final_prompt(
|
||||
if additional_context:
|
||||
parts.append(f"\n## Additional Context\n{additional_context}")
|
||||
|
||||
# Tool call history
|
||||
# Tool call history — include as many entries as fit within the token budget,
|
||||
# preferring the most recent calls (they tend to be the most targeted).
|
||||
if context_history:
|
||||
parts.append("\n## Retrieved Data (synthesize and reason from this data)")
|
||||
for entry in context_history:
|
||||
token_budget = int(max_context_tokens * _FINAL_PROMPT_CONTEXT_FRACTION)
|
||||
# Render entries newest-first, then reverse so the prompt reads chronologically.
|
||||
rendered: list[str] = []
|
||||
truncated = False
|
||||
for entry in reversed(context_history):
|
||||
tool = entry["tool"]
|
||||
output = entry["output"]
|
||||
# Format as proper JSON for LLM readability
|
||||
try:
|
||||
output_str = json.dumps(output, indent=2, default=str)
|
||||
except (TypeError, ValueError):
|
||||
output_str = str(output)
|
||||
parts.append(f"\n### From {tool}:\n```json\n{output_str}\n```")
|
||||
block = f"\n### From {tool}:\n```json\n{output_str}\n```"
|
||||
block_tokens = len(_TIKTOKEN_ENCODING.encode(block))
|
||||
if block_tokens > token_budget:
|
||||
truncated = True
|
||||
break
|
||||
rendered.append(block)
|
||||
token_budget -= block_tokens
|
||||
for block in reversed(rendered):
|
||||
parts.append(block)
|
||||
if truncated:
|
||||
parts.append("\n*Note: Some earlier tool results were omitted to stay within the context window.*")
|
||||
else:
|
||||
parts.append("\n## Retrieved Data\nNo data was retrieved.")
|
||||
|
||||
@@ -510,4 +505,6 @@ CRITICAL: Output ONLY the final synthesized answer. Do NOT include:
|
||||
- Meta-commentary about what you're doing ("I'll search...", "Let me analyze...")
|
||||
- Explanations of your reasoning process
|
||||
- Descriptions of your approach
|
||||
Just provide the direct answer with proper markdown formatting."""
|
||||
Just provide the direct answer with proper markdown formatting.
|
||||
|
||||
CRITICAL: This is a NON-CONVERSATIONAL system. NEVER ask follow-up questions, offer to search again, suggest alternatives, or end with anything like "Would you like me to..." or "Let me know if...". The user cannot reply. Your answer must be complete and self-contained."""
|
||||
|
||||
@@ -9,7 +9,7 @@ Implements hierarchical retrieval:
|
||||
|
||||
import logging
|
||||
import uuid
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from datetime import datetime, timezone
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
if TYPE_CHECKING:
|
||||
@@ -20,9 +20,6 @@ if TYPE_CHECKING:
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Observation is considered stale if not updated in this many days
|
||||
STALE_THRESHOLD_DAYS = 7
|
||||
|
||||
|
||||
async def tool_search_mental_models(
|
||||
conn: "Connection",
|
||||
@@ -33,6 +30,7 @@ async def tool_search_mental_models(
|
||||
tags: list[str] | None = None,
|
||||
tags_match: str = "any",
|
||||
exclude_ids: list[str] | None = None,
|
||||
pending_consolidation: int = 0,
|
||||
) -> dict[str, Any]:
|
||||
"""
|
||||
Search user-curated mental models by semantic similarity.
|
||||
@@ -87,7 +85,6 @@ async def tool_search_mental_models(
|
||||
*params,
|
||||
)
|
||||
|
||||
now = datetime.now(timezone.utc)
|
||||
mental_models = []
|
||||
|
||||
for row in rows:
|
||||
@@ -95,11 +92,10 @@ async def tool_search_mental_models(
|
||||
if last_refreshed_at and last_refreshed_at.tzinfo is None:
|
||||
last_refreshed_at = last_refreshed_at.replace(tzinfo=timezone.utc)
|
||||
|
||||
# Calculate freshness
|
||||
is_stale = False
|
||||
if last_refreshed_at:
|
||||
age = now - last_refreshed_at
|
||||
is_stale = age > timedelta(days=STALE_THRESHOLD_DAYS)
|
||||
# A mental model is stale when there are memories that haven't been consolidated yet —
|
||||
# the same signal used for observations staleness.
|
||||
is_stale = pending_consolidation > 0
|
||||
staleness_reason = f"{pending_consolidation} memories pending consolidation" if is_stale else None
|
||||
|
||||
mental_models.append(
|
||||
{
|
||||
@@ -110,6 +106,7 @@ async def tool_search_mental_models(
|
||||
"relevance": round(row["relevance"], 4),
|
||||
"updated_at": last_refreshed_at.isoformat() if last_refreshed_at else None,
|
||||
"is_stale": is_stale,
|
||||
"staleness_reason": staleness_reason,
|
||||
}
|
||||
)
|
||||
|
||||
@@ -132,7 +129,7 @@ async def tool_search_observations(
|
||||
pending_consolidation: int = 0,
|
||||
) -> dict[str, Any]:
|
||||
"""
|
||||
Search consolidated observations using recall with include_observations.
|
||||
Search consolidated observations using recall with include_source_facts.
|
||||
|
||||
Observations are auto-generated from memories. Returns freshness info
|
||||
so the agent knows if it should also verify with recall().
|
||||
@@ -149,72 +146,24 @@ async def tool_search_observations(
|
||||
pending_consolidation: Number of memories waiting to be consolidated
|
||||
|
||||
Returns:
|
||||
Dict with matching observations including freshness info
|
||||
Dict with matching observations including freshness info and source memories
|
||||
"""
|
||||
from ..memory_engine import fq_table
|
||||
|
||||
# Use recall to search observations (they come back in results field when fact_type=["observation"])
|
||||
result = await memory_engine.recall_async(
|
||||
bank_id=bank_id,
|
||||
query=query,
|
||||
fact_type=["observation"], # Only retrieve observations
|
||||
max_tokens=max_tokens, # Token budget controls how many observations are returned
|
||||
fact_type=["observation"],
|
||||
max_tokens=max_tokens,
|
||||
enable_trace=False,
|
||||
request_context=request_context,
|
||||
tags=tags,
|
||||
tags_match=tags_match,
|
||||
include_source_facts=True,
|
||||
max_source_facts_tokens=-1, # No token limit — include all source facts
|
||||
_connection_budget=1,
|
||||
_quiet=True,
|
||||
)
|
||||
|
||||
observations = []
|
||||
|
||||
# When fact_type=["observation"], results come back in `results` field as MemoryFact objects
|
||||
# We need to fetch additional fields (proof_count, source_memory_ids) from the database
|
||||
if result.results:
|
||||
obs_ids = [m.id for m in result.results]
|
||||
|
||||
# Fetch proof_count and source_memory_ids for these observations
|
||||
pool = await memory_engine._get_pool()
|
||||
async with pool.acquire() as conn:
|
||||
obs_rows = await conn.fetch(
|
||||
f"""
|
||||
SELECT id, proof_count, source_memory_ids
|
||||
FROM {fq_table("memory_units")}
|
||||
WHERE id = ANY($1::uuid[])
|
||||
""",
|
||||
obs_ids,
|
||||
)
|
||||
obs_data = {str(row["id"]): row for row in obs_rows}
|
||||
|
||||
for m in result.results:
|
||||
# Get additional data from DB lookup
|
||||
extra = obs_data.get(m.id, {})
|
||||
proof_count = extra.get("proof_count", 1) if extra else 1
|
||||
source_ids = extra.get("source_memory_ids", []) if extra else []
|
||||
# Convert UUIDs to strings
|
||||
source_memory_ids = [str(sid) for sid in (source_ids or [])]
|
||||
|
||||
# Determine staleness
|
||||
is_stale = False
|
||||
staleness_reason = None
|
||||
if pending_consolidation > 0:
|
||||
is_stale = True
|
||||
staleness_reason = f"{pending_consolidation} memories pending consolidation"
|
||||
|
||||
observations.append(
|
||||
{
|
||||
"id": str(m.id),
|
||||
"text": m.text,
|
||||
"proof_count": proof_count,
|
||||
"source_memory_ids": source_memory_ids,
|
||||
"tags": m.tags or [],
|
||||
"is_stale": is_stale,
|
||||
"staleness_reason": staleness_reason,
|
||||
}
|
||||
)
|
||||
|
||||
# Return freshness info (more understandable than raw pending_consolidation count)
|
||||
is_stale = pending_consolidation > 0
|
||||
if pending_consolidation == 0:
|
||||
freshness = "up_to_date"
|
||||
elif pending_consolidation < 10:
|
||||
@@ -224,8 +173,10 @@ async def tool_search_observations(
|
||||
|
||||
return {
|
||||
"query": query,
|
||||
"count": len(observations),
|
||||
"observations": observations,
|
||||
"count": len(result.results),
|
||||
"observations": [m.model_dump() for m in result.results],
|
||||
"source_facts": {k: v.model_dump() for k, v in (result.source_facts or {}).items()},
|
||||
"is_stale": is_stale,
|
||||
"freshness": freshness,
|
||||
}
|
||||
|
||||
@@ -236,10 +187,10 @@ async def tool_recall(
|
||||
query: str,
|
||||
request_context: "RequestContext",
|
||||
max_tokens: int = 2048,
|
||||
max_results: int = 50,
|
||||
tags: list[str] | None = None,
|
||||
tags_match: str = "any",
|
||||
connection_budget: int = 1,
|
||||
max_chunk_tokens: int = 1000,
|
||||
) -> dict[str, Any]:
|
||||
"""
|
||||
Search memories using TEMPR retrieval.
|
||||
@@ -253,18 +204,19 @@ async def tool_recall(
|
||||
query: Search query
|
||||
request_context: Request context for authentication
|
||||
max_tokens: Maximum tokens for results (default 2048)
|
||||
max_results: Maximum number of results
|
||||
tags: Filter by tags (includes untagged memories)
|
||||
tags_match: How to match tags - "any" (OR), "all" (AND), or "exact"
|
||||
connection_budget: Max DB connections for this recall (default 1 for internal ops)
|
||||
max_chunk_tokens: Maximum tokens for raw source chunk text (default 1000, always included)
|
||||
|
||||
Returns:
|
||||
Dict with list of matching memories
|
||||
Dict with list of matching memories including raw chunk text
|
||||
"""
|
||||
include_chunks = True
|
||||
result = await memory_engine.recall_async(
|
||||
bank_id=bank_id,
|
||||
query=query,
|
||||
fact_type=["experience", "world"], # Exclude opinions and observations
|
||||
fact_type=["experience", "world"],
|
||||
max_tokens=max_tokens,
|
||||
enable_trace=False,
|
||||
request_context=request_context,
|
||||
@@ -272,24 +224,14 @@ async def tool_recall(
|
||||
tags_match=tags_match,
|
||||
_connection_budget=connection_budget,
|
||||
_quiet=True, # Suppress logging for internal operations
|
||||
include_chunks=include_chunks,
|
||||
max_chunk_tokens=max_chunk_tokens,
|
||||
)
|
||||
|
||||
memories = []
|
||||
for m in result.results[:max_results]:
|
||||
memories.append(
|
||||
{
|
||||
"id": str(m.id),
|
||||
"text": m.text,
|
||||
"type": m.fact_type,
|
||||
"entities": m.entities or [],
|
||||
"occurred": m.occurred_start, # Already ISO format string
|
||||
}
|
||||
)
|
||||
|
||||
return {
|
||||
"query": query,
|
||||
"count": len(memories),
|
||||
"memories": memories,
|
||||
"memories": [m.model_dump() for m in result.results],
|
||||
"chunks": {k: v.model_dump() for k, v in (result.chunks or {}).items()},
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -47,7 +47,8 @@ TOOL_SEARCH_OBSERVATIONS = {
|
||||
"description": (
|
||||
"Search consolidated observations (auto-generated knowledge). These are automatically "
|
||||
"synthesized from memories. Returns observations with freshness info (updated_at, is_stale). "
|
||||
"If an observation is STALE, you should ALSO use recall() to verify with current facts."
|
||||
"If an observation is STALE, you should ALSO use recall() to verify with current facts. "
|
||||
"IMPORTANT: If search_mental_models is available, you MUST call it FIRST before using this tool."
|
||||
),
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
@@ -95,6 +96,10 @@ TOOL_RECALL = {
|
||||
"type": "integer",
|
||||
"description": "Optional limit on result size (default 2048). Use higher values for broader searches.",
|
||||
},
|
||||
"max_chunk_tokens": {
|
||||
"type": "integer",
|
||||
"description": "Maximum tokens for raw source chunk text included alongside each memory fact (default 1000, min 1000). Chunks provide the surrounding context the fact was extracted from. Increase for broader context.",
|
||||
},
|
||||
},
|
||||
"required": ["reason", "query"],
|
||||
},
|
||||
@@ -139,7 +144,7 @@ TOOL_DONE_ANSWER = {
|
||||
"properties": {
|
||||
"answer": {
|
||||
"type": "string",
|
||||
"description": "Your response as well-formatted markdown. Use headers, lists, bold/italic, and code blocks for clarity. NEVER include memory IDs, UUIDs, or 'Memory references' in this text - put IDs only in memory_ids array.",
|
||||
"description": "Your response as well-formatted markdown. Use headers, lists, bold/italic, and code blocks for clarity. NEVER include memory IDs, UUIDs, or 'Memory references' in this text - put IDs only in memory_ids array. LANGUAGE: By default, write in the SAME language as the user's question. However, if a language directive in the system prompt specifies a different language, follow that directive instead.",
|
||||
},
|
||||
"memory_ids": {
|
||||
"type": "array",
|
||||
@@ -190,7 +195,11 @@ def _build_done_tool_with_directives(directive_rules: list[str]) -> dict:
|
||||
"properties": {
|
||||
"answer": {
|
||||
"type": "string",
|
||||
"description": "Your response as well-formatted markdown. Use headers, lists, bold/italic, and code blocks for clarity. NEVER include memory IDs, UUIDs, or 'Memory references' in this text - put IDs only in memory_ids array.",
|
||||
"description": (
|
||||
"Your response as well-formatted markdown. Use headers, lists, bold/italic, and code blocks for clarity. "
|
||||
"NEVER include memory IDs, UUIDs, or 'Memory references' in this text - put IDs only in memory_ids array. "
|
||||
f"MANDATORY: Your answer MUST comply with ALL directives:\n{rules_list}"
|
||||
),
|
||||
},
|
||||
"memory_ids": {
|
||||
"type": "array",
|
||||
|
||||
@@ -159,6 +159,10 @@ class MemoryFact(BaseModel):
|
||||
None, description="ID of the chunk this fact was extracted from (format: bank_id_document_id_chunk_index)"
|
||||
)
|
||||
tags: list[str] | None = Field(None, description="Visibility scope tags associated with this fact")
|
||||
source_fact_ids: list[str] | None = Field(
|
||||
None,
|
||||
description="IDs of source facts this observation was derived from (observation type only, when source_facts is enabled)",
|
||||
)
|
||||
|
||||
|
||||
class ChunkInfo(BaseModel):
|
||||
@@ -226,6 +230,9 @@ class RecallResult(BaseModel):
|
||||
chunks: dict[str, ChunkInfo] | None = Field(
|
||||
None, description="Chunks for facts, keyed by '{document_id}_{chunk_index}'"
|
||||
)
|
||||
source_facts: dict[str, MemoryFact] | None = Field(
|
||||
None, description="Source facts for observation-type results, keyed by fact ID"
|
||||
)
|
||||
|
||||
|
||||
class ReflectResult(BaseModel):
|
||||
|
||||
@@ -27,11 +27,21 @@ def augment_texts_with_dates(facts: list[ExtractedFact], format_date_fn) -> list
|
||||
"""
|
||||
augmented_texts = []
|
||||
for fact in facts:
|
||||
# Use occurred_start as the representative date
|
||||
# Use occurred_start as the representative date, fall back to mentioned_at
|
||||
fact_date = fact.occurred_start or fact.mentioned_at
|
||||
readable_date = format_date_fn(fact_date)
|
||||
# Augment text with date for embedding (but store original text in DB)
|
||||
augmented_text = f"{fact.fact_text} (happened in {readable_date})"
|
||||
# Augment text with date and entity names for embedding (but store original text in DB)
|
||||
# Entity names (including key:value labels) improve retrieval without polluting stored content
|
||||
if fact_date is not None:
|
||||
readable_date = format_date_fn(fact_date)
|
||||
if fact.occurred_end and fact.occurred_end != fact.occurred_start:
|
||||
readable_end = format_date_fn(fact.occurred_end)
|
||||
augmented_text = f"{fact.fact_text} (happened from {readable_date} to {readable_end})"
|
||||
else:
|
||||
augmented_text = f"{fact.fact_text} (happened in {readable_date})"
|
||||
else:
|
||||
augmented_text = fact.fact_text
|
||||
if fact.entities:
|
||||
augmented_text = f"{augmented_text} [{', '.join(fact.entities)}]"
|
||||
augmented_texts.append(augmented_text)
|
||||
return augmented_texts
|
||||
|
||||
|
||||
@@ -0,0 +1,194 @@
|
||||
"""
|
||||
Entity labels models and helpers for retain pipeline.
|
||||
|
||||
Defines a controlled vocabulary of key:value classification labels
|
||||
(e.g., 'pedagogy:scaffolding', 'interest:active') that are extracted
|
||||
at retain time and stored as entities.
|
||||
"""
|
||||
|
||||
from typing import Literal
|
||||
|
||||
from pydantic import BaseModel, Field, create_model
|
||||
|
||||
|
||||
class LabelValue(BaseModel):
|
||||
"""A single allowed value for a label group."""
|
||||
|
||||
value: str
|
||||
description: str = ""
|
||||
|
||||
|
||||
class LabelGroup(BaseModel):
|
||||
"""A label group (dimension) with its type and allowed values."""
|
||||
|
||||
key: str
|
||||
description: str = ""
|
||||
type: Literal["value", "multi-values", "text"] = "value"
|
||||
optional: bool = True
|
||||
tag: bool = False
|
||||
values: list[LabelValue] = []
|
||||
|
||||
|
||||
class EntityLabelsConfig(BaseModel):
|
||||
"""Entity labels configuration for a bank (controlled vocabulary)."""
|
||||
|
||||
attributes: list[LabelGroup] = []
|
||||
|
||||
|
||||
def parse_entity_labels(raw: dict | list | None) -> EntityLabelsConfig | None:
|
||||
"""
|
||||
Parse raw entity labels config into EntityLabelsConfig.
|
||||
|
||||
Accepts:
|
||||
- None → returns None
|
||||
- list → list of attribute dicts (each may use legacy free_values/multi_value or new type field)
|
||||
- dict → {attributes: [...]}
|
||||
|
||||
Legacy migration (backward-compat):
|
||||
- free_values=True → type="text"
|
||||
- multi_value=True → type="multi-values"
|
||||
- neither / free_values=False → type="value"
|
||||
|
||||
Args:
|
||||
raw: Raw entity labels config from bank config
|
||||
|
||||
Returns:
|
||||
EntityLabelsConfig or None if raw is None/empty
|
||||
"""
|
||||
if raw is None:
|
||||
return None
|
||||
|
||||
if isinstance(raw, list):
|
||||
if not raw:
|
||||
return None
|
||||
attributes = [LabelGroup.model_validate(_migrate_label_group(a)) for a in raw]
|
||||
return EntityLabelsConfig(attributes=attributes)
|
||||
|
||||
if isinstance(raw, dict):
|
||||
attrs_raw = raw.get("attributes", [])
|
||||
if not attrs_raw:
|
||||
return None
|
||||
attributes = [LabelGroup.model_validate(_migrate_label_group(a)) for a in attrs_raw]
|
||||
return EntityLabelsConfig(attributes=attributes)
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def _migrate_label_group(raw: dict) -> dict:
|
||||
"""Migrate legacy free_values/multi_value fields to the new type field."""
|
||||
if not isinstance(raw, dict) or "type" in raw:
|
||||
return raw
|
||||
patched = dict(raw)
|
||||
if patched.get("free_values"):
|
||||
patched["type"] = "text"
|
||||
elif patched.get("multi_value"):
|
||||
patched["type"] = "multi-values"
|
||||
else:
|
||||
patched["type"] = "value"
|
||||
# Remove legacy keys so Pydantic doesn't error on unknown fields
|
||||
patched.pop("free_values", None)
|
||||
patched.pop("multi_value", None)
|
||||
return patched
|
||||
|
||||
|
||||
def build_labels_model(labels_cfg: EntityLabelsConfig) -> type[BaseModel] | None:
|
||||
"""
|
||||
Build a dynamic Pydantic model for structured label extraction.
|
||||
|
||||
Each LabelGroup becomes a typed field based on its type:
|
||||
- type="text" → str | None (always optional)
|
||||
- type="value", optional=True → Literal["v1","v2"] | None
|
||||
- type="value", optional=False → Literal["v1","v2"] (required)
|
||||
- type="multi-values" → list[Literal["v1","v2"]]
|
||||
|
||||
Args:
|
||||
labels_cfg: Parsed EntityLabelsConfig
|
||||
|
||||
Returns:
|
||||
Dynamic Pydantic model class, or None if no groups defined
|
||||
"""
|
||||
fields: dict = {}
|
||||
for group in labels_cfg.attributes:
|
||||
if not group.key:
|
||||
continue
|
||||
description = group.description or group.key
|
||||
|
||||
if group.type == "text":
|
||||
# Free-form: any string value accepted, always optional
|
||||
fields[group.key] = (str | None, Field(default=None, description=description))
|
||||
else:
|
||||
# Enum-constrained: must have defined values
|
||||
if not group.values:
|
||||
continue
|
||||
values = tuple(v.value for v in group.values if v.value)
|
||||
if not values:
|
||||
continue
|
||||
# Literal[("v1", "v2")] is equivalent to Literal["v1", "v2"] in Python 3.11+
|
||||
literal_type = Literal[values] # type: ignore[valid-type]
|
||||
if group.type == "multi-values":
|
||||
fields[group.key] = (
|
||||
list[literal_type], # type: ignore[valid-type]
|
||||
Field(default_factory=list, description=description),
|
||||
)
|
||||
elif group.optional:
|
||||
fields[group.key] = (
|
||||
literal_type | None, # type: ignore[valid-type]
|
||||
Field(default=None, description=description),
|
||||
)
|
||||
else:
|
||||
fields[group.key] = (
|
||||
literal_type, # type: ignore[valid-type]
|
||||
Field(description=description),
|
||||
)
|
||||
|
||||
if not fields:
|
||||
return None
|
||||
|
||||
return create_model("Labels", **fields)
|
||||
|
||||
|
||||
def is_label_entity(text: str, labels_cfg: EntityLabelsConfig, labels_lookup: set[str]) -> bool:
|
||||
"""
|
||||
Return True if entity text belongs to any configured label group.
|
||||
|
||||
For enum groups: checks the pre-built lookup set.
|
||||
For text groups: checks that the text starts with a known key prefix.
|
||||
"""
|
||||
if text.lower() in labels_lookup:
|
||||
return True
|
||||
for group in labels_cfg.attributes:
|
||||
if group.type == "text" and group.key and text.lower().startswith(f"{group.key.lower()}:"):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def build_labels_lookup(labels_cfg: EntityLabelsConfig | list | None) -> set[str]:
|
||||
"""
|
||||
Build a set of valid 'key:value' label strings (lowercase) for fast lookup.
|
||||
|
||||
Accepts either EntityLabelsConfig or raw list/None for backwards compatibility.
|
||||
|
||||
Args:
|
||||
labels_cfg: EntityLabelsConfig, raw list of attribute dicts, or None
|
||||
|
||||
Returns:
|
||||
Set of lowercase 'key:value' strings
|
||||
"""
|
||||
if labels_cfg is None:
|
||||
return set()
|
||||
|
||||
# Accept raw list/dict for backwards compatibility
|
||||
if not isinstance(labels_cfg, EntityLabelsConfig):
|
||||
parsed = parse_entity_labels(labels_cfg)
|
||||
if parsed is None:
|
||||
return set()
|
||||
labels_cfg = parsed
|
||||
|
||||
valid = set()
|
||||
for group in labels_cfg.attributes:
|
||||
if group.type == "text":
|
||||
continue # No fixed vocabulary — all values accepted in post-processing
|
||||
for v in group.values:
|
||||
if group.key and v.value:
|
||||
valid.add(f"{group.key}:{v.value}".lower())
|
||||
return valid
|
||||
@@ -20,6 +20,7 @@ async def process_entities_batch(
|
||||
facts: list[ProcessedFact],
|
||||
log_buffer: list[str] = None,
|
||||
user_entities_per_content: dict[int, list[dict]] = None,
|
||||
entity_labels: list | None = None,
|
||||
) -> list[EntityLink]:
|
||||
"""
|
||||
Process entities for all facts and create entity links.
|
||||
@@ -90,6 +91,7 @@ async def process_entities_batch(
|
||||
fact_dates,
|
||||
entities_per_fact,
|
||||
log_buffer, # Pass log_buffer for detailed logging
|
||||
entity_labels=entity_labels,
|
||||
)
|
||||
|
||||
return entity_links
|
||||
|
||||
@@ -10,23 +10,31 @@ import json
|
||||
import logging
|
||||
import re
|
||||
from datetime import datetime, timedelta
|
||||
from typing import Literal
|
||||
from typing import Literal, cast
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field, field_validator
|
||||
from pydantic import BaseModel, ConfigDict, Field, create_model, field_validator
|
||||
|
||||
from ...config import get_config
|
||||
from ..llm_wrapper import LLMConfig, OutputTooLongError
|
||||
from ..response_models import TokenUsage
|
||||
from .entity_labels import (
|
||||
EntityLabelsConfig,
|
||||
build_labels_lookup,
|
||||
build_labels_model,
|
||||
is_label_entity,
|
||||
parse_entity_labels,
|
||||
)
|
||||
|
||||
|
||||
def _infer_temporal_date(fact_text: str, event_date: datetime) -> str | None:
|
||||
def _infer_temporal_date(fact_text: str, event_date: datetime | None) -> str | None:
|
||||
"""
|
||||
Infer a temporal date from fact text when LLM didn't provide occurred_start.
|
||||
|
||||
This is a fallback for when the LLM fails to extract temporal information
|
||||
from relative time expressions like "last night", "yesterday", etc.
|
||||
"""
|
||||
import re
|
||||
if event_date is None:
|
||||
return None
|
||||
|
||||
fact_lower = fact_text.lower()
|
||||
|
||||
@@ -102,7 +110,6 @@ class Fact(BaseModel):
|
||||
# Optional temporal fields
|
||||
occurred_start: str | None = None
|
||||
occurred_end: str | None = None
|
||||
mentioned_at: str | None = None
|
||||
|
||||
# Optional location field
|
||||
where: str | None = Field(
|
||||
@@ -440,11 +447,9 @@ def _chunk_conversation(turns: list[dict], max_chars: int) -> list[str]:
|
||||
# Uses {extraction_guidelines} placeholder for mode-specific instructions
|
||||
_BASE_FACT_EXTRACTION_PROMPT = """Extract SIGNIFICANT facts from text. Be SELECTIVE - only extract facts worth remembering long-term.
|
||||
|
||||
LANGUAGE REQUIREMENT: Detect the language of the input text. All extracted facts, entity names, descriptions, and other output MUST be in the SAME language as the input. Do not translate to another language.
|
||||
LANGUAGE: MANDATORY — Detect the language of the input text and produce ALL output in that EXACT same language. You are STRICTLY FORBIDDEN from translating or switching to any other language. Every single word of your output must be in the same language as the input. Do NOT output in a different language under any circumstance.
|
||||
|
||||
{fact_types_instruction}
|
||||
|
||||
{extraction_guidelines}
|
||||
{retain_mission_section}{extraction_guidelines}
|
||||
|
||||
══════════════════════════════════════════════════════════════════════════
|
||||
FACT FORMAT - BE CONCISE
|
||||
@@ -483,7 +488,9 @@ TEMPORAL HANDLING
|
||||
══════════════════════════════════════════════════════════════════════════
|
||||
|
||||
Use "Event Date" from input as reference for relative dates.
|
||||
- "yesterday" relative to Event Date, not today
|
||||
- CRITICAL: Convert ALL relative temporal expressions to absolute dates in the fact text itself.
|
||||
"yesterday" → write the resolved date (e.g. "on November 12, 2024"), NOT the word "yesterday"
|
||||
"last night", "this morning", "today", "tonight" → convert to the resolved absolute date
|
||||
- For events: set occurred_start AND occurred_end (same for point events)
|
||||
- For conversation facts: NO occurred dates
|
||||
|
||||
@@ -521,7 +528,7 @@ CONSOLIDATE related statements into ONE fact when possible."""
|
||||
_CONCISE_EXAMPLES = """
|
||||
|
||||
══════════════════════════════════════════════════════════════════════════
|
||||
EXAMPLES
|
||||
EXAMPLES (shown in English for illustration; for non-English input, ALL output values MUST be in the input language)
|
||||
══════════════════════════════════════════════════════════════════════════
|
||||
|
||||
Example 1 - Selective extraction (Event Date: June 10, 2024):
|
||||
@@ -549,16 +556,16 @@ about experiences ARE important to remember, even if they seem small (e.g., how
|
||||
tasted, how someone looked, how loud music was). Extract these if they characterize
|
||||
an experience or person."""
|
||||
|
||||
# Assembled concise prompt (backward compatible - exact same output as before)
|
||||
# Assembled concise prompt
|
||||
CONCISE_FACT_EXTRACTION_PROMPT = _BASE_FACT_EXTRACTION_PROMPT.format(
|
||||
fact_types_instruction="{fact_types_instruction}",
|
||||
retain_mission_section="{retain_mission_section}",
|
||||
extraction_guidelines=_CONCISE_GUIDELINES,
|
||||
examples=_CONCISE_EXAMPLES,
|
||||
)
|
||||
|
||||
# Custom prompt uses same base but without examples
|
||||
CUSTOM_FACT_EXTRACTION_PROMPT = _BASE_FACT_EXTRACTION_PROMPT.format(
|
||||
fact_types_instruction="{fact_types_instruction}",
|
||||
retain_mission_section="{retain_mission_section}",
|
||||
extraction_guidelines="{custom_instructions}",
|
||||
examples="", # No examples for custom mode
|
||||
)
|
||||
@@ -567,10 +574,7 @@ CUSTOM_FACT_EXTRACTION_PROMPT = _BASE_FACT_EXTRACTION_PROMPT.format(
|
||||
# Verbose extraction prompt - detailed, comprehensive facts (legacy mode)
|
||||
VERBOSE_FACT_EXTRACTION_PROMPT = """Extract facts from text into structured format with FIVE required dimensions - BE EXTREMELY DETAILED.
|
||||
|
||||
LANGUAGE REQUIREMENT: Detect the language of the input text. All extracted facts, entity names, descriptions,
|
||||
and other output MUST be in the SAME language as the input. Do not translate to English if the input is in another language.
|
||||
|
||||
{fact_types_instruction}
|
||||
LANGUAGE: MANDATORY — Detect the language of the input text and produce ALL output in that EXACT same language. You are STRICTLY FORBIDDEN from translating or switching to any other language. Every single word of your output must be in the same language as the input. Do NOT output in a different language under any circumstance.
|
||||
|
||||
══════════════════════════════════════════════════════════════════════════
|
||||
FACT FORMAT - ALL FIVE DIMENSIONS REQUIRED - MAXIMUM VERBOSITY
|
||||
@@ -695,59 +699,185 @@ Example: "Lost job → couldn't pay rent → moved apartment"
|
||||
- Fact 2: Moved apartment, causal_relations: [{target_index: 1, relation_type: "caused_by"}]"""
|
||||
|
||||
|
||||
def _build_labels_prompt_section(labels_cfg: EntityLabelsConfig | list | None, free_form_entities: bool = True) -> str:
|
||||
"""Build the entity labels classification section for the extraction prompt."""
|
||||
if labels_cfg is None:
|
||||
return ""
|
||||
|
||||
# Accept raw list for backwards compatibility
|
||||
if isinstance(labels_cfg, list):
|
||||
if not labels_cfg:
|
||||
return ""
|
||||
labels_cfg = parse_entity_labels(labels_cfg)
|
||||
if labels_cfg is None:
|
||||
return ""
|
||||
|
||||
if not labels_cfg.attributes:
|
||||
return ""
|
||||
|
||||
if free_form_entities:
|
||||
entities_instruction = "Classify each fact using the structured 'labels' field below. Continue extracting regular named entities in the 'entities' field."
|
||||
else:
|
||||
entities_instruction = "Classify each fact using the structured 'labels' field below. Do NOT add regular named entities — labels-only mode."
|
||||
|
||||
lines = [
|
||||
"\n\n══════════════════════════════════════════════════════════════════════════",
|
||||
"ENTITY LABELS - CLASSIFICATION ATTRIBUTES",
|
||||
"══════════════════════════════════════════════════════════════════════════",
|
||||
"",
|
||||
entities_instruction,
|
||||
"",
|
||||
"For each fact, fill the 'labels' object. Each field is a label group:",
|
||||
"",
|
||||
]
|
||||
|
||||
for attr in labels_cfg.attributes:
|
||||
if attr.type == "text":
|
||||
# Free-text: no predefined values — LLM writes any relevant string or null
|
||||
lines.append(f"- {attr.key} (free text or null): {attr.description}")
|
||||
else:
|
||||
mode = "multi-value (list)" if attr.type == "multi-values" else "single value or null"
|
||||
lines.append(f"- {attr.key} ({mode}): {attr.description}")
|
||||
for v in attr.values:
|
||||
desc = f" — {v.description}" if v.description else ""
|
||||
lines.append(f' • "{v.value}"{desc}')
|
||||
lines.append("")
|
||||
|
||||
lines.append("Only assign labels when clearly applicable. Leave null/empty if the fact does not match.")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def _build_extraction_prompt_and_schema(config) -> tuple[str, type]:
|
||||
"""
|
||||
Build extraction prompt and response schema based on config.
|
||||
|
||||
When a taxonomy is configured, dynamically builds a Pydantic model with a
|
||||
typed `taxonomy_entities` field using an Enum built from valid taxonomy values.
|
||||
This enables JSON schema enforcement for structured outputs.
|
||||
|
||||
Returns:
|
||||
Tuple of (prompt, response_schema)
|
||||
"""
|
||||
fact_types_instruction = "Extract ONLY 'world' and 'assistant' type facts."
|
||||
extraction_mode = config.retain_extraction_mode
|
||||
extract_causal_links = config.retain_extract_causal_links
|
||||
|
||||
# Build retain_mission section if set - injected before the mode-specific guidelines
|
||||
retain_mission = getattr(config, "retain_mission", None)
|
||||
if retain_mission:
|
||||
retain_mission_section = (
|
||||
f"══════════════════════════════════════════════════════════════════════════\n"
|
||||
f"FOCUS — What to retain for this bank\n"
|
||||
f"══════════════════════════════════════════════════════════════════════════\n\n"
|
||||
f"{retain_mission}\n\n"
|
||||
)
|
||||
else:
|
||||
retain_mission_section = ""
|
||||
|
||||
# Select base prompt based on extraction mode
|
||||
if extraction_mode == "custom":
|
||||
if not config.retain_custom_instructions:
|
||||
base_prompt = CONCISE_FACT_EXTRACTION_PROMPT
|
||||
prompt = base_prompt.format(fact_types_instruction=fact_types_instruction)
|
||||
prompt = base_prompt.format(
|
||||
retain_mission_section=retain_mission_section,
|
||||
)
|
||||
else:
|
||||
base_prompt = CUSTOM_FACT_EXTRACTION_PROMPT
|
||||
prompt = base_prompt.format(
|
||||
fact_types_instruction=fact_types_instruction,
|
||||
retain_mission_section=retain_mission_section,
|
||||
custom_instructions=config.retain_custom_instructions,
|
||||
)
|
||||
elif extraction_mode == "verbose":
|
||||
base_prompt = VERBOSE_FACT_EXTRACTION_PROMPT
|
||||
prompt = base_prompt.format(fact_types_instruction=fact_types_instruction)
|
||||
prompt = VERBOSE_FACT_EXTRACTION_PROMPT
|
||||
else:
|
||||
base_prompt = CONCISE_FACT_EXTRACTION_PROMPT
|
||||
prompt = base_prompt.format(fact_types_instruction=fact_types_instruction)
|
||||
prompt = base_prompt.format(
|
||||
retain_mission_section=retain_mission_section,
|
||||
)
|
||||
|
||||
# Add causal relationships section if enabled
|
||||
if extract_causal_links:
|
||||
prompt = prompt + CAUSAL_RELATIONSHIPS_SECTION
|
||||
response_schema = FactExtractionResponseVerbose if extraction_mode == "verbose" else FactExtractionResponse
|
||||
base_fact_class = ExtractedFactVerbose if extraction_mode == "verbose" else ExtractedFact
|
||||
base_response_class = FactExtractionResponseVerbose if extraction_mode == "verbose" else FactExtractionResponse
|
||||
else:
|
||||
response_schema = FactExtractionResponseNoCausal
|
||||
base_fact_class = ExtractedFactNoCausal
|
||||
base_response_class = FactExtractionResponseNoCausal
|
||||
|
||||
# Add entity labels section if configured and build dynamic schema
|
||||
entity_labels_raw = getattr(config, "entity_labels", None)
|
||||
labels_cfg = parse_entity_labels(entity_labels_raw)
|
||||
free_form_entities = getattr(config, "entities_allow_free_form", True)
|
||||
labels_section = _build_labels_prompt_section(labels_cfg, free_form_entities)
|
||||
if labels_section:
|
||||
prompt = prompt + labels_section
|
||||
|
||||
response_schema = base_response_class
|
||||
|
||||
if labels_cfg and labels_cfg.attributes:
|
||||
LabelsModel = build_labels_model(labels_cfg)
|
||||
if LabelsModel is not None:
|
||||
dynamic_fields: dict = {
|
||||
"labels": (
|
||||
LabelsModel,
|
||||
Field(
|
||||
description="Classification labels for this fact. Fill each applicable field; leave others null/empty."
|
||||
),
|
||||
)
|
||||
}
|
||||
if not free_form_entities:
|
||||
dynamic_fields["entities"] = (
|
||||
list[Entity] | None,
|
||||
Field(default=None, description="Leave empty — labels-only mode"),
|
||||
)
|
||||
# Inherit parent's required fields and add 'labels' so it appears in the JSON schema
|
||||
# required array (the base class json_schema_extra overrides required entirely)
|
||||
base_extra = base_fact_class.model_config.get("json_schema_extra")
|
||||
base_required = cast(dict, base_extra).get("required", []) if isinstance(base_extra, dict) else []
|
||||
DynamicFact = create_model(
|
||||
"LabelsFact",
|
||||
__base__=base_fact_class,
|
||||
__config__=ConfigDict(
|
||||
json_schema_mode="validation",
|
||||
json_schema_extra={"required": [*base_required, "labels"]},
|
||||
),
|
||||
**dynamic_fields,
|
||||
)
|
||||
DynamicResponse = create_model("LabelsResponse", facts=(list[DynamicFact], ...)) # type: ignore[valid-type]
|
||||
response_schema = DynamicResponse
|
||||
|
||||
return prompt, response_schema
|
||||
|
||||
|
||||
def _build_user_message(chunk: str, chunk_index: int, total_chunks: int, event_date: datetime, context: str) -> str:
|
||||
def _build_user_message(
|
||||
chunk: str,
|
||||
chunk_index: int,
|
||||
total_chunks: int,
|
||||
event_date: datetime | None,
|
||||
context: str,
|
||||
metadata: dict[str, str] | None = None,
|
||||
) -> str:
|
||||
"""Build user message for fact extraction."""
|
||||
from .orchestrator import parse_datetime_flexible
|
||||
|
||||
sanitized_chunk = _sanitize_text(chunk)
|
||||
sanitized_context = _sanitize_text(context) if context else "none"
|
||||
event_date = parse_datetime_flexible(event_date)
|
||||
event_date_formatted = event_date.strftime("%A, %B %d, %Y")
|
||||
|
||||
if event_date is not None:
|
||||
event_date = parse_datetime_flexible(event_date)
|
||||
event_date_str = f"{event_date.strftime('%A, %B %d, %Y')} ({event_date.isoformat()})"
|
||||
else:
|
||||
event_date_str = "Unknown"
|
||||
|
||||
metadata_section = ""
|
||||
if metadata:
|
||||
metadata_lines = "\n".join(f" {k}: {v}" for k, v in metadata.items())
|
||||
metadata_section = f"\nMetadata:\n{metadata_lines}"
|
||||
|
||||
return f"""Extract facts from the following text chunk.
|
||||
|
||||
Chunk: {chunk_index + 1}/{total_chunks}
|
||||
Event Date: {event_date_formatted} ({event_date.isoformat()})
|
||||
Context: {sanitized_context}
|
||||
Event Date: {event_date_str}
|
||||
Context: {sanitized_context}{metadata_section}
|
||||
|
||||
Text:
|
||||
{sanitized_chunk}"""
|
||||
@@ -784,11 +914,12 @@ async def _extract_facts_from_chunk(
|
||||
chunk: str,
|
||||
chunk_index: int,
|
||||
total_chunks: int,
|
||||
event_date: datetime,
|
||||
event_date: datetime | None,
|
||||
context: str,
|
||||
llm_config: "LLMConfig",
|
||||
config,
|
||||
agent_name: str = None,
|
||||
metadata: dict[str, str] | None = None,
|
||||
) -> tuple[list[dict[str, str]], TokenUsage]:
|
||||
"""
|
||||
Extract facts from a single chunk (internal helper for parallel processing).
|
||||
@@ -810,7 +941,7 @@ async def _extract_facts_from_chunk(
|
||||
extract_causal_links = config.retain_extract_causal_links
|
||||
|
||||
# Build user message using helper function
|
||||
user_message = _build_user_message(chunk, chunk_index, total_chunks, event_date, context)
|
||||
user_message = _build_user_message(chunk, chunk_index, total_chunks, event_date, context, metadata)
|
||||
|
||||
# Retry logic for JSON validation errors
|
||||
max_retries = 2
|
||||
@@ -969,9 +1100,9 @@ async def _extract_facts_from_chunk(
|
||||
# Add entities if present (validate as Entity objects)
|
||||
# LLM sometimes returns strings instead of {"text": "..."} format
|
||||
entities = get_value("entities")
|
||||
validated_entities = []
|
||||
if entities:
|
||||
# Validate and normalize each entity
|
||||
validated_entities = []
|
||||
for ent in entities:
|
||||
if isinstance(ent, str):
|
||||
# Normalize string to Entity object
|
||||
@@ -981,8 +1112,48 @@ async def _extract_facts_from_chunk(
|
||||
validated_entities.append(Entity.model_validate(ent))
|
||||
except Exception as e:
|
||||
logger.warning(f"Invalid entity {ent}: {e}")
|
||||
if validated_entities:
|
||||
fact_data["entities"] = validated_entities
|
||||
|
||||
# Post-process label entities from structured labels object
|
||||
entity_labels_raw = getattr(config, "entity_labels", None)
|
||||
labels_cfg = parse_entity_labels(entity_labels_raw)
|
||||
free_form_entities = getattr(config, "entities_allow_free_form", True)
|
||||
if labels_cfg and labels_cfg.attributes:
|
||||
labels_lookup = build_labels_lookup(labels_cfg)
|
||||
labels_data = llm_fact.get("labels") or {}
|
||||
if isinstance(labels_data, dict):
|
||||
existing_texts_lower = {e.text.lower() for e in validated_entities}
|
||||
for group in labels_cfg.attributes:
|
||||
value = labels_data.get(group.key)
|
||||
if not value:
|
||||
continue
|
||||
values_list = value if isinstance(value, list) else [value]
|
||||
for v in values_list:
|
||||
if not isinstance(v, str) or not v.strip() or v.lower() in ("none", "null", "n/a"):
|
||||
continue
|
||||
label_str = f"{group.key}:{v.strip()}"
|
||||
if group.type == "text":
|
||||
if label_str.lower() not in existing_texts_lower:
|
||||
validated_entities.append(Entity(text=label_str))
|
||||
existing_texts_lower.add(label_str.lower())
|
||||
elif (
|
||||
label_str.lower() in labels_lookup and label_str.lower() not in existing_texts_lower
|
||||
):
|
||||
validated_entities.append(Entity(text=label_str))
|
||||
existing_texts_lower.add(label_str.lower())
|
||||
else:
|
||||
logger.warning(f"Label '{label_str}' not in valid label values, skipping")
|
||||
|
||||
# In labels-only mode, keep only label entities
|
||||
if not free_form_entities:
|
||||
validated_entities = [
|
||||
e for e in validated_entities if is_label_entity(e.text, labels_cfg, labels_lookup)
|
||||
]
|
||||
elif not free_form_entities:
|
||||
# No labels but free_form disabled: clear all entities
|
||||
validated_entities = []
|
||||
|
||||
if validated_entities:
|
||||
fact_data["entities"] = validated_entities
|
||||
|
||||
# Add per-fact causal relations (only if enabled in config)
|
||||
if extract_causal_links:
|
||||
@@ -1021,8 +1192,9 @@ async def _extract_facts_from_chunk(
|
||||
if validated_relations:
|
||||
fact_data["causal_relations"] = validated_relations
|
||||
|
||||
# Always set mentioned_at to the event_date (when the conversation/document occurred)
|
||||
fact_data["mentioned_at"] = event_date.isoformat()
|
||||
# Set mentioned_at to the event_date (when the conversation/document occurred),
|
||||
# or None when the caller opted into no timestamp.
|
||||
fact_data["mentioned_at"] = event_date.isoformat() if event_date is not None else None
|
||||
|
||||
# Build Fact model instance
|
||||
try:
|
||||
@@ -1085,11 +1257,12 @@ async def _extract_facts_with_auto_split(
|
||||
chunk: str,
|
||||
chunk_index: int,
|
||||
total_chunks: int,
|
||||
event_date: datetime,
|
||||
event_date: datetime | None,
|
||||
context: str,
|
||||
llm_config: LLMConfig,
|
||||
config,
|
||||
agent_name: str = None,
|
||||
metadata: dict[str, str] | None = None,
|
||||
) -> tuple[list[dict[str, str]], TokenUsage]:
|
||||
"""
|
||||
Extract facts from a chunk with automatic splitting if output exceeds token limits.
|
||||
@@ -1106,6 +1279,7 @@ async def _extract_facts_with_auto_split(
|
||||
llm_config: LLM configuration to use
|
||||
config: Resolved HindsightConfig for this bank
|
||||
agent_name: Optional agent name (memory owner)
|
||||
metadata: Optional document metadata key-value pairs
|
||||
|
||||
Returns:
|
||||
Tuple of (facts list, token usage) extracted from the chunk (possibly from sub-chunks)
|
||||
@@ -1125,6 +1299,7 @@ async def _extract_facts_with_auto_split(
|
||||
llm_config=llm_config,
|
||||
config=config,
|
||||
agent_name=agent_name,
|
||||
metadata=metadata,
|
||||
)
|
||||
except OutputTooLongError:
|
||||
# Output exceeded token limits - split the chunk in half and retry
|
||||
@@ -1170,6 +1345,7 @@ async def _extract_facts_with_auto_split(
|
||||
llm_config=llm_config,
|
||||
config=config,
|
||||
agent_name=agent_name,
|
||||
metadata=metadata,
|
||||
),
|
||||
_extract_facts_with_auto_split(
|
||||
chunk=second_half,
|
||||
@@ -1180,6 +1356,7 @@ async def _extract_facts_with_auto_split(
|
||||
llm_config=llm_config,
|
||||
config=config,
|
||||
agent_name=agent_name,
|
||||
metadata=metadata,
|
||||
),
|
||||
]
|
||||
|
||||
@@ -1199,11 +1376,12 @@ async def _extract_facts_with_auto_split(
|
||||
|
||||
async def extract_facts_from_text(
|
||||
text: str,
|
||||
event_date: datetime,
|
||||
event_date: datetime | None,
|
||||
llm_config: LLMConfig,
|
||||
agent_name: str,
|
||||
config,
|
||||
context: str = "",
|
||||
metadata: dict[str, str] | None = None,
|
||||
) -> tuple[list[Fact], list[tuple[str, int]], TokenUsage]:
|
||||
"""
|
||||
Extract semantic facts from conversational or narrative text using LLM.
|
||||
@@ -1221,6 +1399,7 @@ async def extract_facts_from_text(
|
||||
agent_name: Agent name (memory owner)
|
||||
config: Resolved HindsightConfig for this bank
|
||||
context: Context about the conversation/document
|
||||
metadata: Optional document metadata key-value pairs
|
||||
|
||||
Returns:
|
||||
Tuple of (facts, chunks, usage) where:
|
||||
@@ -1248,6 +1427,7 @@ async def extract_facts_from_text(
|
||||
llm_config=llm_config,
|
||||
config=config,
|
||||
agent_name=agent_name,
|
||||
metadata=metadata,
|
||||
)
|
||||
for i, chunk in enumerate(chunks)
|
||||
]
|
||||
@@ -1274,8 +1454,8 @@ from .types import ExtractedFact as ExtractedFactType
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Each fact gets 10 seconds offset to preserve ordering within a document
|
||||
SECONDS_PER_FACT = 10
|
||||
# Each fact gets 10ms offset to preserve ordering within a document
|
||||
SECONDS_PER_FACT = 0.01
|
||||
|
||||
|
||||
async def extract_facts_from_contents_batch_api(
|
||||
@@ -1357,7 +1537,7 @@ async def extract_facts_from_contents_batch_api(
|
||||
|
||||
# Build user message using helper function
|
||||
user_message = _build_user_message(
|
||||
chunk, chunk_index_in_content, len(chunks), item.event_date, item.context
|
||||
chunk, chunk_index_in_content, len(chunks), item.event_date, item.context, item.metadata or None
|
||||
)
|
||||
|
||||
# Build request body using helper function
|
||||
@@ -1569,8 +1749,8 @@ async def extract_facts_from_contents_batch_api(
|
||||
|
||||
# Entities
|
||||
entities = get_value("entities")
|
||||
validated_entities = []
|
||||
if entities:
|
||||
validated_entities = []
|
||||
for ent in entities:
|
||||
if isinstance(ent, str):
|
||||
validated_entities.append(Entity(text=ent))
|
||||
@@ -1579,8 +1759,45 @@ async def extract_facts_from_contents_batch_api(
|
||||
validated_entities.append(Entity.model_validate(ent))
|
||||
except Exception:
|
||||
pass
|
||||
if validated_entities:
|
||||
fact_data["entities"] = validated_entities
|
||||
|
||||
# Post-process label entities from structured labels object
|
||||
entity_labels_raw = getattr(config, "entity_labels", None)
|
||||
labels_cfg_batch = parse_entity_labels(entity_labels_raw)
|
||||
free_form_entities_batch = getattr(config, "entities_allow_free_form", True)
|
||||
if labels_cfg_batch and labels_cfg_batch.attributes:
|
||||
labels_lookup_batch = build_labels_lookup(labels_cfg_batch)
|
||||
labels_data = llm_fact.get("labels") or {}
|
||||
if isinstance(labels_data, dict):
|
||||
existing_texts_lower = {e.text.lower() for e in validated_entities}
|
||||
for group in labels_cfg_batch.attributes:
|
||||
value = labels_data.get(group.key)
|
||||
if not value:
|
||||
continue
|
||||
values_list = value if isinstance(value, list) else [value]
|
||||
for v in values_list:
|
||||
if not isinstance(v, str) or not v.strip() or v.lower() in ("none", "null", "n/a"):
|
||||
continue
|
||||
label_str = f"{group.key}:{v.strip()}"
|
||||
if group.type == "text":
|
||||
if label_str.lower() not in existing_texts_lower:
|
||||
validated_entities.append(Entity(text=label_str))
|
||||
existing_texts_lower.add(label_str.lower())
|
||||
elif (
|
||||
label_str.lower() in labels_lookup_batch
|
||||
and label_str.lower() not in existing_texts_lower
|
||||
):
|
||||
validated_entities.append(Entity(text=label_str))
|
||||
existing_texts_lower.add(label_str.lower())
|
||||
|
||||
if not free_form_entities_batch:
|
||||
validated_entities = [
|
||||
e for e in validated_entities if is_label_entity(e.text, labels_cfg_batch, labels_lookup_batch)
|
||||
]
|
||||
elif not free_form_entities_batch:
|
||||
validated_entities = []
|
||||
|
||||
if validated_entities:
|
||||
fact_data["entities"] = validated_entities
|
||||
|
||||
# Causal relations
|
||||
if extract_causal_links:
|
||||
@@ -1611,8 +1828,9 @@ async def extract_facts_from_contents_batch_api(
|
||||
if validated_relations:
|
||||
fact_data["causal_relations"] = validated_relations
|
||||
|
||||
# Always set mentioned_at
|
||||
fact_data["mentioned_at"] = event_date.isoformat()
|
||||
# Set mentioned_at to the event_date (when the conversation/document occurred),
|
||||
# or None when the caller opted into no timestamp.
|
||||
fact_data["mentioned_at"] = event_date.isoformat() if event_date is not None else None
|
||||
|
||||
try:
|
||||
fact = Fact(fact=combined_text, fact_type=fact_type, **fact_data)
|
||||
@@ -1671,6 +1889,7 @@ async def extract_facts_from_contents_batch_api(
|
||||
mentioned_at=content.event_date,
|
||||
metadata=content.metadata,
|
||||
tags=content.tags,
|
||||
observation_scopes=content.observation_scopes,
|
||||
)
|
||||
|
||||
extracted_facts.append(extracted_fact)
|
||||
@@ -1679,6 +1898,9 @@ async def extract_facts_from_contents_batch_api(
|
||||
# Step 7: Add temporal offsets
|
||||
_add_temporal_offsets(extracted_facts, contents)
|
||||
|
||||
# Step 8: Auto-tag facts from label groups with tag=True
|
||||
_inject_label_tags(extracted_facts, config)
|
||||
|
||||
logger.info(f"Batch API extracted {len(extracted_facts)} facts from {len(all_chunks_info)} chunks")
|
||||
|
||||
return extracted_facts, chunks_metadata, total_usage
|
||||
@@ -1737,6 +1959,7 @@ async def extract_facts_from_contents(
|
||||
llm_config=llm_config,
|
||||
agent_name=agent_name,
|
||||
config=config,
|
||||
metadata=item.metadata or None,
|
||||
)
|
||||
fact_extraction_tasks.append(task)
|
||||
|
||||
@@ -1800,6 +2023,7 @@ async def extract_facts_from_contents(
|
||||
mentioned_at=content.event_date,
|
||||
metadata=content.metadata,
|
||||
tags=content.tags,
|
||||
observation_scopes=content.observation_scopes,
|
||||
)
|
||||
|
||||
extracted_facts.append(extracted_fact)
|
||||
@@ -1809,6 +2033,9 @@ async def extract_facts_from_contents(
|
||||
# Step 4: Add time offsets to preserve ordering within each content
|
||||
_add_temporal_offsets(extracted_facts, contents)
|
||||
|
||||
# Step 5: Auto-tag facts from label groups with tag=True
|
||||
_inject_label_tags(extracted_facts, config)
|
||||
|
||||
return extracted_facts, chunks_metadata, total_usage
|
||||
|
||||
|
||||
@@ -1864,3 +2091,24 @@ def _add_temporal_offsets(facts: list[ExtractedFactType], contents: list[RetainC
|
||||
fact.occurred_end = parse_datetime_flexible(fact.occurred_end) + offset
|
||||
if fact.mentioned_at:
|
||||
fact.mentioned_at = parse_datetime_flexible(fact.mentioned_at) + offset
|
||||
|
||||
|
||||
def _inject_label_tags(facts: list[ExtractedFactType], config) -> None:
|
||||
"""
|
||||
For label groups with tag=True, add extracted key:value label entities
|
||||
to each fact's tags list. Modifies facts in place.
|
||||
|
||||
This lets entity labels double as tags, enabling filtering via the
|
||||
existing tags API without any extra query infrastructure.
|
||||
"""
|
||||
labels_cfg = parse_entity_labels(getattr(config, "entity_labels", None))
|
||||
if not labels_cfg:
|
||||
return
|
||||
tag_group_keys = {g.key.lower() for g in labels_cfg.attributes if g.tag}
|
||||
if not tag_group_keys:
|
||||
return
|
||||
for fact in facts:
|
||||
label_tags = [e for e in fact.entities if ":" in e and e.split(":", 1)[0].lower() in tag_group_keys]
|
||||
if label_tags:
|
||||
existing = set(fact.tags)
|
||||
fact.tags = fact.tags + [t for t in label_tags if t not in existing]
|
||||
|
||||
@@ -47,6 +47,8 @@ async def insert_facts_batch(
|
||||
chunk_ids = []
|
||||
document_ids = []
|
||||
tags_list = []
|
||||
observation_scopes_list = []
|
||||
text_signals_list = []
|
||||
|
||||
for fact in facts:
|
||||
fact_texts.append(_sanitize_text(fact.fact_text))
|
||||
@@ -68,6 +70,19 @@ async def insert_facts_batch(
|
||||
document_ids.append(fact.document_id if fact.document_id else document_id)
|
||||
# Convert tags to JSON string for proper batch insertion (PostgreSQL unnest doesn't handle 2D arrays well)
|
||||
tags_list.append(json.dumps(fact.tags if fact.tags else []))
|
||||
# observation_scopes: stored as JSONB (string or 2D array), None if not provided
|
||||
observation_scopes_list.append(
|
||||
json.dumps(fact.observation_scopes) if fact.observation_scopes is not None else None
|
||||
)
|
||||
# Build text_signals: entity names + date tokens for enriched BM25 indexing
|
||||
signal_parts = []
|
||||
if fact.entities:
|
||||
signal_parts.extend(e.name for e in fact.entities)
|
||||
if fact.occurred_start:
|
||||
signal_parts.append(fact.occurred_start.strftime("%B %-d %Y"))
|
||||
if fact.occurred_end and fact.occurred_end != fact.occurred_start:
|
||||
signal_parts.append(fact.occurred_end.strftime("%B %-d %Y"))
|
||||
text_signals_list.append(" ".join(signal_parts) if signal_parts else None)
|
||||
|
||||
# Batch insert all facts
|
||||
# Note: tags are passed as JSON strings and converted back to varchar[] via jsonb_array_elements_text + array_agg
|
||||
@@ -75,16 +90,19 @@ async def insert_facts_batch(
|
||||
config = get_config()
|
||||
if config.text_search_extension == "vchord":
|
||||
# VectorChord: manually tokenize and insert search_vector
|
||||
# text_signals (entity names etc.) are included in the tokenize input for enriched BM25
|
||||
query = f"""
|
||||
WITH input_data AS (
|
||||
SELECT * FROM unnest(
|
||||
$2::text[], $3::vector[], $4::timestamptz[], $5::timestamptz[], $6::timestamptz[], $7::timestamptz[],
|
||||
$8::text[], $9::text[], $10::float[], $11::jsonb[], $12::text[], $13::text[], $14::jsonb[]
|
||||
$8::text[], $9::text[], $10::float[], $11::jsonb[], $12::text[], $13::text[], $14::jsonb[], $15::jsonb[], $16::text[]
|
||||
) AS t(text, embedding, event_date, occurred_start, occurred_end, mentioned_at,
|
||||
context, fact_type, confidence_score, metadata, chunk_id, document_id, tags_json)
|
||||
context, fact_type, confidence_score, metadata, chunk_id, document_id, tags_json,
|
||||
observation_scopes_json, text_signals)
|
||||
)
|
||||
INSERT INTO {fq_table("memory_units")} (bank_id, text, embedding, event_date, occurred_start, occurred_end, mentioned_at,
|
||||
context, fact_type, confidence_score, metadata, chunk_id, document_id, tags, search_vector)
|
||||
context, fact_type, confidence_score, metadata, chunk_id, document_id, tags,
|
||||
observation_scopes, text_signals, search_vector)
|
||||
SELECT
|
||||
$1,
|
||||
text, embedding, event_date, occurred_start, occurred_end, mentioned_at,
|
||||
@@ -93,23 +111,30 @@ async def insert_facts_batch(
|
||||
(SELECT array_agg(elem) FROM jsonb_array_elements_text(tags_json) AS elem),
|
||||
'{{}}'::varchar[]
|
||||
),
|
||||
tokenize(COALESCE(text, '') || ' ' || COALESCE(context, ''), 'llmlingua2')::bm25_catalog.bm25vector
|
||||
observation_scopes_json,
|
||||
text_signals,
|
||||
tokenize(
|
||||
COALESCE(text, '') || ' ' || COALESCE(context, '') || ' ' || COALESCE(text_signals, ''),
|
||||
'llmlingua2'
|
||||
)::bm25_catalog.bm25vector
|
||||
FROM input_data
|
||||
RETURNING id
|
||||
"""
|
||||
else: # native or pg_textsearch
|
||||
# Native PostgreSQL: search_vector is GENERATED ALWAYS, don't include it
|
||||
# Native PostgreSQL: search_vector is GENERATED ALWAYS (expression includes text_signals), don't include it
|
||||
# pg_textsearch: indexes operate on base columns directly, don't populate search_vector
|
||||
query = f"""
|
||||
WITH input_data AS (
|
||||
SELECT * FROM unnest(
|
||||
$2::text[], $3::vector[], $4::timestamptz[], $5::timestamptz[], $6::timestamptz[], $7::timestamptz[],
|
||||
$8::text[], $9::text[], $10::float[], $11::jsonb[], $12::text[], $13::text[], $14::jsonb[]
|
||||
$8::text[], $9::text[], $10::float[], $11::jsonb[], $12::text[], $13::text[], $14::jsonb[], $15::jsonb[], $16::text[]
|
||||
) AS t(text, embedding, event_date, occurred_start, occurred_end, mentioned_at,
|
||||
context, fact_type, confidence_score, metadata, chunk_id, document_id, tags_json)
|
||||
context, fact_type, confidence_score, metadata, chunk_id, document_id, tags_json,
|
||||
observation_scopes_json, text_signals)
|
||||
)
|
||||
INSERT INTO {fq_table("memory_units")} (bank_id, text, embedding, event_date, occurred_start, occurred_end, mentioned_at,
|
||||
context, fact_type, confidence_score, metadata, chunk_id, document_id, tags)
|
||||
context, fact_type, confidence_score, metadata, chunk_id, document_id, tags,
|
||||
observation_scopes, text_signals)
|
||||
SELECT
|
||||
$1,
|
||||
text, embedding, event_date, occurred_start, occurred_end, mentioned_at,
|
||||
@@ -117,7 +142,9 @@ async def insert_facts_batch(
|
||||
COALESCE(
|
||||
(SELECT array_agg(elem) FROM jsonb_array_elements_text(tags_json) AS elem),
|
||||
'{{}}'::varchar[]
|
||||
)
|
||||
),
|
||||
observation_scopes_json,
|
||||
text_signals
|
||||
FROM input_data
|
||||
RETURNING id
|
||||
"""
|
||||
@@ -138,6 +165,8 @@ async def insert_facts_batch(
|
||||
chunk_ids,
|
||||
document_ids,
|
||||
tags_list,
|
||||
observation_scopes_list,
|
||||
text_signals_list,
|
||||
)
|
||||
|
||||
unit_ids = [str(row["id"]) for row in results]
|
||||
|
||||
@@ -47,6 +47,9 @@ def compute_temporal_links(
|
||||
|
||||
links = []
|
||||
for unit_id, unit_event_date in new_units.items():
|
||||
# Units without event_date can't form temporal links
|
||||
if unit_event_date is None:
|
||||
continue
|
||||
# Normalize unit_event_date for consistent comparison
|
||||
unit_event_date_norm = _normalize_datetime(unit_event_date)
|
||||
|
||||
@@ -96,7 +99,11 @@ def compute_temporal_query_bounds(
|
||||
return None, None
|
||||
|
||||
# Normalize all dates to be timezone-aware to avoid comparison issues
|
||||
all_dates = [_normalize_datetime(d) for d in new_units.values()]
|
||||
# Filter out None values — units without event_date can't form temporal links
|
||||
all_dates = [_normalize_datetime(d) for d in new_units.values() if d is not None]
|
||||
|
||||
if not all_dates:
|
||||
return None, None
|
||||
|
||||
try:
|
||||
min_date = min(all_dates) - timedelta(hours=time_window_hours)
|
||||
@@ -143,6 +150,7 @@ async def extract_entities_batch_optimized(
|
||||
fact_dates: list,
|
||||
llm_entities: list[list[dict]],
|
||||
log_buffer: list[str] = None,
|
||||
entity_labels: list | None = None,
|
||||
) -> list[tuple]:
|
||||
"""
|
||||
Process LLM-extracted entities for ALL facts in batch.
|
||||
@@ -232,6 +240,7 @@ async def extract_entities_batch_optimized(
|
||||
context=context,
|
||||
unit_event_date=None, # Not used when per-entity dates provided
|
||||
conn=conn, # Use main transaction connection
|
||||
entity_labels=entity_labels,
|
||||
)
|
||||
|
||||
_log(
|
||||
@@ -432,20 +441,23 @@ async def create_temporal_links_batch_per_fact(
|
||||
min_date, max_date = compute_temporal_query_bounds(new_units, time_window_hours)
|
||||
|
||||
fetch_neighbors_start = time_mod.time()
|
||||
all_candidates = await conn.fetch(
|
||||
f"""
|
||||
SELECT id, event_date
|
||||
FROM {fq_table("memory_units")}
|
||||
WHERE bank_id = $1
|
||||
AND event_date BETWEEN $2 AND $3
|
||||
AND id::text != ALL($4)
|
||||
ORDER BY event_date DESC
|
||||
""",
|
||||
bank_id,
|
||||
min_date,
|
||||
max_date,
|
||||
unit_ids,
|
||||
)
|
||||
if min_date is not None and max_date is not None:
|
||||
all_candidates = await conn.fetch(
|
||||
f"""
|
||||
SELECT id, event_date
|
||||
FROM {fq_table("memory_units")}
|
||||
WHERE bank_id = $1
|
||||
AND event_date BETWEEN $2 AND $3
|
||||
AND id::text != ALL($4)
|
||||
ORDER BY event_date DESC
|
||||
""",
|
||||
bank_id,
|
||||
min_date,
|
||||
max_date,
|
||||
unit_ids,
|
||||
)
|
||||
else:
|
||||
all_candidates = []
|
||||
_log(
|
||||
log_buffer,
|
||||
f" [7.2] Fetch {len(all_candidates)} candidate neighbors (1 query): {time_mod.time() - fetch_neighbors_start:.3f}s",
|
||||
@@ -460,11 +472,15 @@ async def create_temporal_links_batch_per_fact(
|
||||
# Convert new_units dict to candidate format for within-batch linking
|
||||
new_unit_items = list(new_units.items())
|
||||
for i, (unit_id, event_date) in enumerate(new_unit_items):
|
||||
if event_date is None:
|
||||
continue # Skip units without event_date for temporal linking
|
||||
unit_event_date_norm = _normalize_datetime(event_date)
|
||||
|
||||
# Compare with other new units (only those after this one to avoid duplicates)
|
||||
for j in range(i + 1, len(new_unit_items)):
|
||||
other_id, other_event_date = new_unit_items[j]
|
||||
if other_event_date is None:
|
||||
continue # Skip units without event_date
|
||||
other_event_date_norm = _normalize_datetime(other_event_date)
|
||||
|
||||
# Check if within time window
|
||||
|
||||
@@ -128,12 +128,14 @@ async def retain_batch(
|
||||
item_tags = item.get("tags", []) or []
|
||||
merged_tags = list(set(item_tags + (document_tags or [])))
|
||||
|
||||
# Handle event_date: parse flexibly (handles both datetime objects and ISO strings)
|
||||
event_date_value = item.get("event_date")
|
||||
if event_date_value:
|
||||
event_date_value = parse_datetime_flexible(event_date_value)
|
||||
# Handle event_date: distinguish "not provided" (default to now) from
|
||||
# "explicitly None" (caller opted into no timestamp).
|
||||
if "event_date" in item and item["event_date"] is None:
|
||||
event_date_value = None # Caller explicitly signalled "unknown date"
|
||||
elif item.get("event_date"):
|
||||
event_date_value = parse_datetime_flexible(item["event_date"])
|
||||
else:
|
||||
event_date_value = utcnow()
|
||||
event_date_value = utcnow() # Backward-compatible default
|
||||
|
||||
content = RetainContent(
|
||||
content=item["content"],
|
||||
@@ -142,6 +144,7 @@ async def retain_batch(
|
||||
metadata=item.get("metadata", {}),
|
||||
entities=item.get("entities", []),
|
||||
tags=merged_tags,
|
||||
observation_scopes=item.get("observation_scopes"),
|
||||
)
|
||||
contents.append(content)
|
||||
|
||||
@@ -156,13 +159,22 @@ async def retain_batch(
|
||||
)
|
||||
|
||||
if not extracted_facts:
|
||||
# Still need to create document if document_id was provided
|
||||
# Still need to create document if document_id was provided or chunks exist
|
||||
from collections import defaultdict
|
||||
|
||||
docs_tracked = 0
|
||||
async with acquire_with_retry(pool) as conn:
|
||||
async with conn.transaction():
|
||||
await fact_storage.ensure_bank_exists(conn, bank_id)
|
||||
|
||||
# Handle document tracking even with no facts
|
||||
# Group contents by document_id (consistent with normal path)
|
||||
contents_by_doc_early = defaultdict(list)
|
||||
for idx, content_dict in enumerate(contents_dicts):
|
||||
doc_id = content_dict.get("document_id")
|
||||
contents_by_doc_early[doc_id].append((idx, content_dict))
|
||||
|
||||
if document_id:
|
||||
# Legacy: single document_id parameter
|
||||
combined_content = "\n".join([c.get("content", "") for c in contents_dicts])
|
||||
# Collect tags from all content items and merge with document_tags
|
||||
all_tags = set(document_tags or [])
|
||||
@@ -187,45 +199,57 @@ async def retain_batch(
|
||||
await fact_storage.handle_document_tracking(
|
||||
conn, bank_id, document_id, combined_content, is_first_batch, retain_params, merged_tags
|
||||
)
|
||||
docs_tracked += 1
|
||||
else:
|
||||
# Check for per-item document_ids
|
||||
from collections import defaultdict
|
||||
# Handle per-item document_ids and/or chunks (mirrors normal path logic)
|
||||
has_any_doc_ids = any(item.get("document_id") for item in contents_dicts)
|
||||
|
||||
contents_by_doc = defaultdict(list)
|
||||
for idx, content_dict in enumerate(contents_dicts):
|
||||
doc_id = content_dict.get("document_id")
|
||||
if doc_id:
|
||||
contents_by_doc[doc_id].append((idx, content_dict))
|
||||
if has_any_doc_ids or chunks:
|
||||
for original_doc_id, doc_contents in contents_by_doc_early.items():
|
||||
should_create_doc = (original_doc_id is not None) or chunks
|
||||
if not should_create_doc:
|
||||
continue
|
||||
|
||||
for doc_id, doc_contents in contents_by_doc.items():
|
||||
combined_content = "\n".join([c.get("content", "") for _, c in doc_contents])
|
||||
# Collect tags from all content items for this document and merge with document_tags
|
||||
all_tags = set(document_tags or [])
|
||||
for _, item in doc_contents:
|
||||
item_tags = item.get("tags", []) or []
|
||||
all_tags.update(item_tags)
|
||||
merged_tags = list(all_tags)
|
||||
actual_doc_id = original_doc_id
|
||||
if actual_doc_id is None:
|
||||
# No document_id but have chunks - generate one
|
||||
actual_doc_id = str(uuid.uuid4())
|
||||
|
||||
retain_params = {}
|
||||
if doc_contents:
|
||||
first_item = doc_contents[0][1]
|
||||
if first_item.get("context"):
|
||||
retain_params["context"] = first_item["context"]
|
||||
if first_item.get("event_date"):
|
||||
retain_params["event_date"] = (
|
||||
first_item["event_date"].isoformat()
|
||||
if hasattr(first_item["event_date"], "isoformat")
|
||||
else str(first_item["event_date"])
|
||||
)
|
||||
if first_item.get("metadata"):
|
||||
retain_params["metadata"] = first_item["metadata"]
|
||||
await fact_storage.handle_document_tracking(
|
||||
conn, bank_id, doc_id, combined_content, is_first_batch, retain_params, merged_tags
|
||||
)
|
||||
combined_content = "\n".join([c.get("content", "") for _, c in doc_contents])
|
||||
all_tags = set(document_tags or [])
|
||||
for _, item in doc_contents:
|
||||
item_tags = item.get("tags", []) or []
|
||||
all_tags.update(item_tags)
|
||||
merged_tags = list(all_tags)
|
||||
|
||||
retain_params = {}
|
||||
if doc_contents:
|
||||
first_item = doc_contents[0][1]
|
||||
if first_item.get("context"):
|
||||
retain_params["context"] = first_item["context"]
|
||||
if first_item.get("event_date"):
|
||||
retain_params["event_date"] = (
|
||||
first_item["event_date"].isoformat()
|
||||
if hasattr(first_item["event_date"], "isoformat")
|
||||
else str(first_item["event_date"])
|
||||
)
|
||||
if first_item.get("metadata"):
|
||||
retain_params["metadata"] = first_item["metadata"]
|
||||
await fact_storage.handle_document_tracking(
|
||||
conn,
|
||||
bank_id,
|
||||
actual_doc_id,
|
||||
combined_content,
|
||||
is_first_batch,
|
||||
retain_params,
|
||||
merged_tags,
|
||||
)
|
||||
docs_tracked += 1
|
||||
|
||||
total_time = time.time() - start_time
|
||||
doc_status = f"{docs_tracked} document(s) tracked" if docs_tracked > 0 else "no document tracked"
|
||||
logger.info(
|
||||
f"RETAIN_BATCH COMPLETE: 0 facts extracted from {len(contents)} contents in {total_time:.3f}s (document tracked, no facts)"
|
||||
f"RETAIN_BATCH COMPLETE: 0 facts extracted from {len(contents)} contents in {total_time:.3f}s ({doc_status}, no facts)"
|
||||
)
|
||||
return [[] for _ in contents], usage
|
||||
|
||||
@@ -448,6 +472,7 @@ async def retain_batch(
|
||||
non_duplicate_facts,
|
||||
log_buffer,
|
||||
user_entities_per_content=user_entities_per_content,
|
||||
entity_labels=getattr(config, "entity_labels", None),
|
||||
)
|
||||
log_buffer.append(f"[6] Process entities: {len(entity_links)} links in {time.time() - step_start:.3f}s")
|
||||
|
||||
|
||||
@@ -6,8 +6,8 @@ from content input to fact storage.
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import UTC, datetime
|
||||
from typing import TypedDict
|
||||
from datetime import datetime
|
||||
from typing import Literal, TypedDict
|
||||
from uuid import UUID
|
||||
|
||||
|
||||
@@ -22,20 +22,21 @@ class RetainContentDict(TypedDict, total=False):
|
||||
document_id: Document ID for this content item (optional)
|
||||
entities: User-provided entities to merge with extracted entities (optional)
|
||||
tags: Visibility scope tags for this content item (optional)
|
||||
observation_scopes: How to scope observations for consolidation (optional).
|
||||
"per_tag" runs one pass per individual tag; "combined" (default) runs a
|
||||
single pass with all tags; a list[list[str]] specifies exact passes.
|
||||
"""
|
||||
|
||||
content: str # Required
|
||||
context: str
|
||||
event_date: datetime
|
||||
event_date: datetime | None
|
||||
metadata: dict[str, str]
|
||||
document_id: str
|
||||
entities: list[dict[str, str]] # [{"text": "...", "type": "..."}]
|
||||
tags: list[str] # Visibility scope tags
|
||||
|
||||
|
||||
def _now_utc() -> datetime:
|
||||
"""Factory function for default event_date."""
|
||||
return datetime.now(UTC)
|
||||
observation_scopes: (
|
||||
Literal["per_tag", "combined", "all_combinations"] | list[list[str]]
|
||||
) # Observation scopes for consolidation
|
||||
|
||||
|
||||
@dataclass
|
||||
@@ -48,10 +49,13 @@ class RetainContent:
|
||||
|
||||
content: str
|
||||
context: str = ""
|
||||
event_date: datetime = field(default_factory=_now_utc)
|
||||
event_date: datetime | None = None
|
||||
metadata: dict[str, str] = field(default_factory=dict)
|
||||
entities: list[dict[str, str]] = field(default_factory=list) # User-provided entities
|
||||
tags: list[str] = field(default_factory=list) # Visibility scope tags
|
||||
observation_scopes: Literal["per_tag", "combined", "all_combinations"] | list[list[str]] | None = (
|
||||
None # Observation scopes
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
@@ -117,6 +121,9 @@ class ExtractedFact:
|
||||
mentioned_at: datetime | None = None
|
||||
metadata: dict[str, str] = field(default_factory=dict)
|
||||
tags: list[str] = field(default_factory=list) # Visibility scope tags
|
||||
observation_scopes: Literal["per_tag", "combined", "all_combinations"] | list[list[str]] | None = (
|
||||
None # Observation scopes
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
@@ -135,7 +142,7 @@ class ProcessedFact:
|
||||
# Temporal data
|
||||
occurred_start: datetime | None
|
||||
occurred_end: datetime | None
|
||||
mentioned_at: datetime
|
||||
mentioned_at: datetime | None
|
||||
|
||||
# Context and metadata
|
||||
context: str
|
||||
@@ -165,6 +172,9 @@ class ProcessedFact:
|
||||
# Visibility scope tags
|
||||
tags: list[str] = field(default_factory=list)
|
||||
|
||||
# Observation scopes for consolidation
|
||||
observation_scopes: Literal["per_tag", "combined", "all_combinations"] | list[list[str]] | None = None
|
||||
|
||||
@property
|
||||
def is_duplicate(self) -> bool:
|
||||
"""Check if this fact was marked as a duplicate."""
|
||||
@@ -185,12 +195,10 @@ class ProcessedFact:
|
||||
Returns:
|
||||
ProcessedFact ready for storage
|
||||
"""
|
||||
from datetime import datetime
|
||||
|
||||
# Use occurred dates only if explicitly provided by LLM
|
||||
occurred_start = extracted_fact.occurred_start
|
||||
occurred_end = extracted_fact.occurred_end
|
||||
mentioned_at = extracted_fact.mentioned_at or datetime.now(UTC)
|
||||
mentioned_at = extracted_fact.mentioned_at # May be None when caller opted into no timestamp
|
||||
|
||||
# Convert entity strings to EntityRef objects
|
||||
entities = [EntityRef(name=name) for name in extracted_fact.entities]
|
||||
@@ -209,6 +217,7 @@ class ProcessedFact:
|
||||
chunk_id=chunk_id,
|
||||
content_index=extracted_fact.content_index,
|
||||
tags=extracted_fact.tags,
|
||||
observation_scopes=extracted_fact.observation_scopes,
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -162,7 +162,7 @@ class BFSGraphRetriever(GraphRetriever):
|
||||
entry_points = await conn.fetch(
|
||||
f"""
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end,
|
||||
mentioned_at, embedding, fact_type, document_id, chunk_id, tags,
|
||||
mentioned_at, fact_type, document_id, chunk_id, tags,
|
||||
1 - (embedding <=> $1::vector) AS similarity
|
||||
FROM {fq_table("memory_units")}
|
||||
WHERE bank_id = $2
|
||||
@@ -216,7 +216,7 @@ class BFSGraphRetriever(GraphRetriever):
|
||||
neighbors = await conn.fetch(
|
||||
f"""
|
||||
SELECT mu.id, mu.text, mu.context, mu.occurred_start, mu.occurred_end,
|
||||
mu.mentioned_at, mu.embedding, mu.fact_type,
|
||||
mu.mentioned_at, mu.fact_type,
|
||||
mu.document_id, mu.chunk_id, mu.tags,
|
||||
ml.weight, ml.link_type, ml.from_unit_id
|
||||
FROM {fq_table("memory_links")} ml
|
||||
|
||||
@@ -45,7 +45,7 @@ async def _find_semantic_seeds(
|
||||
rows = await conn.fetch(
|
||||
f"""
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end,
|
||||
mentioned_at, embedding, fact_type, document_id, chunk_id, tags,
|
||||
mentioned_at, fact_type, document_id, chunk_id, tags,
|
||||
1 - (embedding <=> $1::vector) AS similarity
|
||||
FROM {fq_table("memory_units")}
|
||||
WHERE bank_id = $2
|
||||
@@ -216,7 +216,7 @@ class LinkExpansionRetriever(GraphRetriever):
|
||||
-- Only exclude the actual seed observations
|
||||
SELECT
|
||||
mu.id, mu.text, mu.context, mu.event_date, mu.occurred_start,
|
||||
mu.occurred_end, mu.mentioned_at, mu.embedding,
|
||||
mu.occurred_end, mu.mentioned_at,
|
||||
mu.fact_type, mu.document_id, mu.chunk_id, mu.tags,
|
||||
COUNT(DISTINCT cs.source_id)::float AS score
|
||||
FROM all_connected_sources cs
|
||||
@@ -239,7 +239,7 @@ class LinkExpansionRetriever(GraphRetriever):
|
||||
f"""
|
||||
SELECT
|
||||
mu.id, mu.text, mu.context, mu.event_date, mu.occurred_start,
|
||||
mu.occurred_end, mu.mentioned_at, mu.embedding,
|
||||
mu.occurred_end, mu.mentioned_at,
|
||||
mu.fact_type, mu.document_id, mu.chunk_id, mu.tags,
|
||||
COUNT(*)::float AS score
|
||||
FROM {fq_table("unit_entities")} seed_ue
|
||||
@@ -264,7 +264,7 @@ class LinkExpansionRetriever(GraphRetriever):
|
||||
f"""
|
||||
SELECT DISTINCT ON (mu.id)
|
||||
mu.id, mu.text, mu.context, mu.event_date, mu.occurred_start,
|
||||
mu.occurred_end, mu.mentioned_at, mu.embedding,
|
||||
mu.occurred_end, mu.mentioned_at,
|
||||
mu.fact_type, mu.document_id, mu.chunk_id, mu.tags,
|
||||
ml.weight + 1.0 AS score
|
||||
FROM {fq_table("memory_links")} ml
|
||||
@@ -291,7 +291,7 @@ class LinkExpansionRetriever(GraphRetriever):
|
||||
WITH outgoing AS (
|
||||
-- Links FROM seeds TO other facts
|
||||
SELECT mu.id, mu.text, mu.context, mu.event_date, mu.occurred_start,
|
||||
mu.occurred_end, mu.mentioned_at, mu.embedding,
|
||||
mu.occurred_end, mu.mentioned_at,
|
||||
mu.fact_type, mu.document_id, mu.chunk_id, mu.tags,
|
||||
ml.weight
|
||||
FROM {fq_table("memory_links")} ml
|
||||
@@ -305,7 +305,7 @@ class LinkExpansionRetriever(GraphRetriever):
|
||||
incoming AS (
|
||||
-- Links FROM other facts TO seeds (reverse direction)
|
||||
SELECT mu.id, mu.text, mu.context, mu.event_date, mu.occurred_start,
|
||||
mu.occurred_end, mu.mentioned_at, mu.embedding,
|
||||
mu.occurred_end, mu.mentioned_at,
|
||||
mu.fact_type, mu.document_id, mu.chunk_id, mu.tags,
|
||||
ml.weight
|
||||
FROM {fq_table("memory_links")} ml
|
||||
@@ -323,12 +323,12 @@ class LinkExpansionRetriever(GraphRetriever):
|
||||
)
|
||||
SELECT DISTINCT ON (id)
|
||||
id, text, context, event_date, occurred_start,
|
||||
occurred_end, mentioned_at, embedding,
|
||||
occurred_end, mentioned_at,
|
||||
fact_type, document_id, chunk_id, tags,
|
||||
(MAX(weight) * 0.5) AS score
|
||||
FROM combined
|
||||
GROUP BY id, text, context, event_date, occurred_start,
|
||||
occurred_end, mentioned_at, embedding,
|
||||
occurred_end, mentioned_at,
|
||||
fact_type, document_id, chunk_id, tags
|
||||
ORDER BY id, score DESC
|
||||
LIMIT $4
|
||||
|
||||
@@ -449,7 +449,7 @@ async def fetch_memory_units_by_ids(
|
||||
rows = await conn.fetch(
|
||||
f"""
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end,
|
||||
mentioned_at, embedding, fact_type, document_id, chunk_id, tags
|
||||
mentioned_at, fact_type, document_id, chunk_id, tags
|
||||
FROM {fq_table("memory_units")}
|
||||
WHERE id = ANY($1::uuid[])
|
||||
AND fact_type = $2
|
||||
|
||||
@@ -127,7 +127,7 @@ async def retrieve_semantic_bm25_combined(
|
||||
results = await conn.fetch(
|
||||
f"""
|
||||
WITH semantic_ranked AS (
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end, mentioned_at, embedding, fact_type, document_id, chunk_id, tags,
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end, mentioned_at, fact_type, document_id, chunk_id, tags,
|
||||
1 - (embedding <=> $1::vector) AS similarity,
|
||||
NULL::float AS bm25_score,
|
||||
'semantic' AS source,
|
||||
@@ -139,7 +139,7 @@ async def retrieve_semantic_bm25_combined(
|
||||
AND (1 - (embedding <=> $1::vector)) >= 0.3
|
||||
{tags_clause}
|
||||
)
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end, mentioned_at, embedding, fact_type, document_id, chunk_id, tags,
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end, mentioned_at, fact_type, document_id, chunk_id, tags,
|
||||
similarity, bm25_score, source
|
||||
FROM semantic_ranked
|
||||
WHERE rn <= $4
|
||||
@@ -194,7 +194,7 @@ async def retrieve_semantic_bm25_combined(
|
||||
# Single query template with backend-specific parts injected
|
||||
query = f"""
|
||||
WITH semantic_ranked AS (
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end, mentioned_at, embedding, fact_type, document_id, chunk_id, tags,
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end, mentioned_at, fact_type, document_id, chunk_id, tags,
|
||||
1 - (embedding <=> $1::vector) AS similarity,
|
||||
NULL::float AS bm25_score,
|
||||
'semantic' AS source,
|
||||
@@ -207,7 +207,7 @@ async def retrieve_semantic_bm25_combined(
|
||||
{tags_clause}
|
||||
),
|
||||
bm25_ranked AS (
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end, mentioned_at, embedding, fact_type, document_id, chunk_id, tags,
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end, mentioned_at, fact_type, document_id, chunk_id, tags,
|
||||
NULL::float AS similarity,
|
||||
{bm25_score_expr} AS bm25_score,
|
||||
'bm25' AS source,
|
||||
@@ -219,12 +219,12 @@ async def retrieve_semantic_bm25_combined(
|
||||
{tags_clause}
|
||||
),
|
||||
semantic AS (
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end, mentioned_at, embedding, fact_type, document_id, chunk_id, tags,
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end, mentioned_at, fact_type, document_id, chunk_id, tags,
|
||||
similarity, bm25_score, source
|
||||
FROM semantic_ranked WHERE rn <= $4
|
||||
),
|
||||
bm25 AS (
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end, mentioned_at, embedding, fact_type, document_id, chunk_id, tags,
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end, mentioned_at, fact_type, document_id, chunk_id, tags,
|
||||
similarity, bm25_score, source
|
||||
FROM bm25_ranked WHERE rn <= $4
|
||||
)
|
||||
@@ -301,7 +301,7 @@ async def retrieve_temporal_combined(
|
||||
entry_points = await conn.fetch(
|
||||
f"""
|
||||
WITH ranked_entries AS (
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end, mentioned_at, embedding, fact_type, document_id, chunk_id, tags,
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end, mentioned_at, fact_type, document_id, chunk_id, tags,
|
||||
1 - (embedding <=> $1::vector) AS similarity,
|
||||
ROW_NUMBER() OVER (PARTITION BY fact_type ORDER BY COALESCE(occurred_start, mentioned_at, occurred_end) DESC, embedding <=> $1::vector) AS rn
|
||||
FROM {fq_table("memory_units")}
|
||||
@@ -321,7 +321,7 @@ async def retrieve_temporal_combined(
|
||||
AND (1 - (embedding <=> $1::vector)) >= $6
|
||||
{tags_clause}
|
||||
)
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end, mentioned_at, embedding, fact_type, document_id, chunk_id, tags, similarity
|
||||
SELECT id, text, context, event_date, occurred_start, occurred_end, mentioned_at, fact_type, document_id, chunk_id, tags, similarity
|
||||
FROM ranked_entries
|
||||
WHERE rn <= 10
|
||||
""",
|
||||
@@ -401,7 +401,7 @@ async def retrieve_temporal_combined(
|
||||
|
||||
neighbors = await conn.fetch(
|
||||
f"""
|
||||
SELECT mu.id, mu.text, mu.context, mu.event_date, mu.occurred_start, mu.occurred_end, mu.mentioned_at, mu.embedding, mu.fact_type, mu.document_id, mu.chunk_id, mu.tags,
|
||||
SELECT mu.id, mu.text, mu.context, mu.event_date, mu.occurred_start, mu.occurred_end, mu.mentioned_at, mu.fact_type, mu.document_id, mu.chunk_id, mu.tags,
|
||||
ml.weight, ml.link_type, ml.from_unit_id,
|
||||
1 - (mu.embedding <=> $1::vector) AS similarity
|
||||
FROM {fq_table("memory_links")} ml
|
||||
|
||||
@@ -46,7 +46,6 @@ class RetrievalResult:
|
||||
mentioned_at: datetime | None = None
|
||||
document_id: str | None = None
|
||||
chunk_id: str | None = None
|
||||
embedding: list[float] | None = None
|
||||
tags: list[str] | None = None # Visibility scope tags
|
||||
|
||||
# Retrieval-specific scores (only one will be set depending on retrieval method)
|
||||
@@ -70,7 +69,6 @@ class RetrievalResult:
|
||||
mentioned_at=row.get("mentioned_at"),
|
||||
document_id=row.get("document_id"),
|
||||
chunk_id=row.get("chunk_id"),
|
||||
embedding=row.get("embedding"),
|
||||
tags=row.get("tags"),
|
||||
similarity=row.get("similarity"),
|
||||
bm25_score=row.get("bm25_score"),
|
||||
@@ -154,7 +152,6 @@ class ScoredResult:
|
||||
"mentioned_at": self.retrieval.mentioned_at,
|
||||
"document_id": self.retrieval.document_id,
|
||||
"chunk_id": self.retrieval.chunk_id,
|
||||
"embedding": self.retrieval.embedding,
|
||||
"tags": self.retrieval.tags,
|
||||
"semantic_similarity": self.retrieval.similarity,
|
||||
"bm25_score": self.retrieval.bm25_score,
|
||||
|
||||
@@ -12,6 +12,7 @@ def create_file_storage(
|
||||
storage_type: str,
|
||||
pool_getter: Callable | None = None,
|
||||
schema: str | None = None,
|
||||
schema_getter: Callable | None = None,
|
||||
**kwargs,
|
||||
) -> FileStorage:
|
||||
"""
|
||||
@@ -20,7 +21,8 @@ def create_file_storage(
|
||||
Args:
|
||||
storage_type: "native" (PostgreSQL BYTEA) or "s3" (S3-compatible object storage)
|
||||
pool_getter: Database pool getter (required for native)
|
||||
schema: Database schema (for native multi-tenant)
|
||||
schema: Static database schema (for native single-tenant)
|
||||
schema_getter: Callable returning current schema at query time (for native multi-tenant)
|
||||
**kwargs: Additional args passed to storage backend
|
||||
|
||||
Returns:
|
||||
@@ -32,7 +34,7 @@ def create_file_storage(
|
||||
if storage_type == "native":
|
||||
if not pool_getter:
|
||||
raise ValueError("pool_getter required for native (PostgreSQL) storage")
|
||||
return PostgreSQLFileStorage(pool_getter=pool_getter, schema=schema)
|
||||
return PostgreSQLFileStorage(pool_getter=pool_getter, schema=schema, schema_getter=schema_getter)
|
||||
elif storage_type == "s3":
|
||||
from ...config import get_config
|
||||
from .s3 import S3FileStorage
|
||||
|
||||
@@ -40,16 +40,30 @@ class PostgreSQLFileStorage(FileStorage):
|
||||
For production/scale, consider S3FileStorage instead.
|
||||
"""
|
||||
|
||||
def __init__(self, pool_getter: Callable[[], "asyncpg.Pool"], schema: str | None = None):
|
||||
def __init__(
|
||||
self,
|
||||
pool_getter: Callable[[], "asyncpg.Pool"],
|
||||
schema: str | None = None,
|
||||
schema_getter: Callable[[], str] | None = None,
|
||||
):
|
||||
"""
|
||||
Initialize PostgreSQL file storage.
|
||||
|
||||
Args:
|
||||
pool_getter: Function that returns asyncpg connection pool
|
||||
schema: Database schema (for multi-tenant support)
|
||||
schema: Static database schema (fallback for single-tenant / tests)
|
||||
schema_getter: Callable returning current schema at query time (for multi-tenant)
|
||||
"""
|
||||
self._pool_getter = pool_getter
|
||||
self._schema = schema
|
||||
self._static_schema = schema
|
||||
self._schema_getter = schema_getter
|
||||
|
||||
@property
|
||||
def _schema(self) -> str | None:
|
||||
"""Resolve schema dynamically per-request when schema_getter is provided."""
|
||||
if self._schema_getter:
|
||||
return self._schema_getter()
|
||||
return self._static_schema
|
||||
|
||||
async def store(
|
||||
self,
|
||||
|
||||
@@ -22,6 +22,11 @@ from hindsight_api.extensions.http import HttpExtension
|
||||
from hindsight_api.extensions.loader import load_extension
|
||||
from hindsight_api.extensions.mcp import MCPExtension
|
||||
from hindsight_api.extensions.operation_validator import (
|
||||
# Bank Management operations
|
||||
BankListContext,
|
||||
BankListResult,
|
||||
BankReadContext,
|
||||
BankWriteContext,
|
||||
# Consolidation operation
|
||||
ConsolidateContext,
|
||||
ConsolidateResult,
|
||||
@@ -70,6 +75,11 @@ __all__ = [
|
||||
"RetainContext",
|
||||
"RetainResult",
|
||||
"ValidationResult",
|
||||
# Operation Validator - Bank Management
|
||||
"BankListContext",
|
||||
"BankListResult",
|
||||
"BankReadContext",
|
||||
"BankWriteContext",
|
||||
# Operation Validator - Consolidation
|
||||
"ConsolidateContext",
|
||||
"ConsolidateResult",
|
||||
|
||||
@@ -200,6 +200,44 @@ class ConsolidateResult:
|
||||
error: str | None = None
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Bank Management Contexts
|
||||
# =============================================================================
|
||||
|
||||
|
||||
@dataclass
|
||||
class BankReadContext:
|
||||
"""Context for a bank read operation validation (pre-operation)."""
|
||||
|
||||
bank_id: str
|
||||
operation: str # "get_bank_profile", "get_bank_stats"
|
||||
request_context: "RequestContext"
|
||||
|
||||
|
||||
@dataclass
|
||||
class BankWriteContext:
|
||||
"""Context for a bank write operation validation (pre-operation)."""
|
||||
|
||||
bank_id: str
|
||||
operation: str # "delete_bank", "update_bank", "update_bank_disposition", "set_bank_mission", "merge_bank_mission", "clear_observations", "clear_observations_for_memory"
|
||||
request_context: "RequestContext"
|
||||
|
||||
|
||||
@dataclass
|
||||
class BankListContext:
|
||||
"""Context for filtering the bank list (post-query)."""
|
||||
|
||||
banks: list[dict]
|
||||
request_context: "RequestContext"
|
||||
|
||||
|
||||
@dataclass
|
||||
class BankListResult:
|
||||
"""Result of filtering the bank list."""
|
||||
|
||||
banks: list[dict]
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Mental Model Contexts
|
||||
# =============================================================================
|
||||
@@ -535,3 +573,63 @@ class OperationValidatorExtension(Extension, ABC):
|
||||
- error: Error message (if failed)
|
||||
"""
|
||||
pass
|
||||
|
||||
# =========================================================================
|
||||
# Bank Management - Validation hooks (optional - override to implement)
|
||||
# =========================================================================
|
||||
|
||||
async def validate_bank_read(self, ctx: BankReadContext) -> ValidationResult:
|
||||
"""
|
||||
Validate a bank read operation before execution.
|
||||
|
||||
Override to implement custom validation logic for bank reads
|
||||
(get_bank_profile, get_bank_stats).
|
||||
|
||||
Args:
|
||||
ctx: Context containing:
|
||||
- bank_id: Bank identifier
|
||||
- operation: Operation name
|
||||
- request_context: Request context with auth info
|
||||
|
||||
Returns:
|
||||
ValidationResult indicating whether the operation is allowed.
|
||||
"""
|
||||
return ValidationResult.accept()
|
||||
|
||||
async def validate_bank_write(self, ctx: BankWriteContext) -> ValidationResult:
|
||||
"""
|
||||
Validate a bank write operation before execution.
|
||||
|
||||
Override to implement custom validation logic for bank writes
|
||||
(delete_bank, update_bank, update_bank_disposition, set_bank_mission,
|
||||
merge_bank_mission, clear_observations, clear_observations_for_memory).
|
||||
|
||||
Args:
|
||||
ctx: Context containing:
|
||||
- bank_id: Bank identifier
|
||||
- operation: Operation name
|
||||
- request_context: Request context with auth info
|
||||
|
||||
Returns:
|
||||
ValidationResult indicating whether the operation is allowed.
|
||||
"""
|
||||
return ValidationResult.accept()
|
||||
|
||||
async def filter_bank_list(self, ctx: BankListContext) -> BankListResult:
|
||||
"""
|
||||
Filter the bank list after querying.
|
||||
|
||||
Unlike validate_* methods, this is a post-query filter that narrows results
|
||||
rather than a gate that blocks the operation.
|
||||
|
||||
Override to implement custom filtering (e.g., restrict to allowed banks).
|
||||
|
||||
Args:
|
||||
ctx: Context containing:
|
||||
- banks: List of bank dicts from the database
|
||||
- request_context: Request context with auth info
|
||||
|
||||
Returns:
|
||||
BankListResult with the filtered list of banks.
|
||||
"""
|
||||
return BankListResult(banks=ctx.banks)
|
||||
|
||||
@@ -231,12 +231,15 @@ def main():
|
||||
reranker_litellm_sdk_api_key=config.reranker_litellm_sdk_api_key,
|
||||
reranker_litellm_sdk_model=config.reranker_litellm_sdk_model,
|
||||
reranker_litellm_sdk_api_base=config.reranker_litellm_sdk_api_base,
|
||||
reranker_zeroentropy_api_key=config.reranker_zeroentropy_api_key,
|
||||
reranker_zeroentropy_model=config.reranker_zeroentropy_model,
|
||||
host=args.host,
|
||||
port=args.port,
|
||||
base_path=config.base_path,
|
||||
log_level=args.log_level,
|
||||
log_format=config.log_format,
|
||||
mcp_enabled=config.mcp_enabled,
|
||||
mcp_enabled_tools=config.mcp_enabled_tools,
|
||||
enable_bank_config_api=config.enable_bank_config_api,
|
||||
graph_retriever=config.graph_retriever,
|
||||
mpfp_top_k_neighbors=config.mpfp_top_k_neighbors,
|
||||
@@ -246,6 +249,7 @@ def main():
|
||||
retain_chunk_size=config.retain_chunk_size,
|
||||
retain_extract_causal_links=config.retain_extract_causal_links,
|
||||
retain_extraction_mode=config.retain_extraction_mode,
|
||||
retain_mission=config.retain_mission,
|
||||
retain_custom_instructions=config.retain_custom_instructions,
|
||||
retain_batch_tokens=config.retain_batch_tokens,
|
||||
retain_batch_enabled=config.retain_batch_enabled,
|
||||
@@ -262,13 +266,19 @@ def main():
|
||||
file_storage_azure_account_name=config.file_storage_azure_account_name,
|
||||
file_storage_azure_account_key=config.file_storage_azure_account_key,
|
||||
file_parser=config.file_parser,
|
||||
file_parser_iris_token=config.file_parser_iris_token,
|
||||
file_parser_iris_org_id=config.file_parser_iris_org_id,
|
||||
file_conversion_max_batch_size_mb=config.file_conversion_max_batch_size_mb,
|
||||
file_conversion_max_batch_size=config.file_conversion_max_batch_size,
|
||||
enable_file_upload_api=config.enable_file_upload_api,
|
||||
file_delete_after_retain=config.file_delete_after_retain,
|
||||
enable_observations=config.enable_observations,
|
||||
consolidation_batch_size=config.consolidation_batch_size,
|
||||
consolidation_llm_batch_size=config.consolidation_llm_batch_size,
|
||||
consolidation_max_tokens=config.consolidation_max_tokens,
|
||||
observations_mission=config.observations_mission,
|
||||
entity_labels=config.entity_labels,
|
||||
entities_allow_free_form=config.entities_allow_free_form,
|
||||
skip_llm_verification=config.skip_llm_verification,
|
||||
lazy_reranker=config.lazy_reranker,
|
||||
run_migrations_on_startup=config.run_migrations_on_startup,
|
||||
@@ -284,6 +294,11 @@ def main():
|
||||
worker_max_slots=config.worker_max_slots,
|
||||
worker_consolidation_max_slots=config.worker_consolidation_max_slots,
|
||||
reflect_max_iterations=config.reflect_max_iterations,
|
||||
reflect_max_context_tokens=config.reflect_max_context_tokens,
|
||||
reflect_mission=config.reflect_mission,
|
||||
disposition_skepticism=config.disposition_skepticism,
|
||||
disposition_literalism=config.disposition_literalism,
|
||||
disposition_empathy=config.disposition_empathy,
|
||||
mental_model_refresh_concurrency=config.mental_model_refresh_concurrency,
|
||||
otel_traces_enabled=config.otel_traces_enabled,
|
||||
otel_exporter_otlp_endpoint=config.otel_exporter_otlp_endpoint,
|
||||
|
||||
@@ -1,8 +1,14 @@
|
||||
"""
|
||||
Local MCP server for use with Claude Code (stdio transport).
|
||||
Local MCP server entry point for use with Claude Code (HTTP transport).
|
||||
|
||||
This runs a fully local Hindsight instance with embedded PostgreSQL (pg0).
|
||||
No external database or server required.
|
||||
This is a thin wrapper around the main hindsight-api server that pre-configures
|
||||
sensible defaults for local use (embedded PostgreSQL via pg0, warning log level).
|
||||
|
||||
The full API runs on localhost:8888. Configure Claude Code's MCP settings:
|
||||
claude mcp add --transport http hindsight http://localhost:8888/mcp/
|
||||
|
||||
Or pinned to a specific bank (single-bank mode):
|
||||
claude mcp add --transport http hindsight http://localhost:8888/mcp/default/
|
||||
|
||||
Run with:
|
||||
hindsight-local-mcp
|
||||
@@ -10,148 +16,24 @@ Run with:
|
||||
Or with uvx:
|
||||
uvx hindsight-api@latest hindsight-local-mcp
|
||||
|
||||
Configure in Claude Code's MCP settings:
|
||||
{
|
||||
"mcpServers": {
|
||||
"hindsight": {
|
||||
"command": "uvx",
|
||||
"args": ["hindsight-api@latest", "hindsight-local-mcp"],
|
||||
"env": {
|
||||
"HINDSIGHT_API_LLM_API_KEY": "your-openai-key"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Environment variables:
|
||||
HINDSIGHT_API_LLM_API_KEY: Required. API key for LLM provider.
|
||||
HINDSIGHT_API_LLM_PROVIDER: Optional. LLM provider (default: "openai").
|
||||
HINDSIGHT_API_LLM_MODEL: Optional. LLM model (default: "gpt-4o-mini").
|
||||
HINDSIGHT_API_MCP_LOCAL_BANK_ID: Optional. Memory bank ID (default: "mcp").
|
||||
HINDSIGHT_API_LOG_LEVEL: Optional. Log level (default: "warning").
|
||||
HINDSIGHT_API_MCP_INSTRUCTIONS: Optional. Additional instructions appended to both retain and recall tools.
|
||||
|
||||
Example custom instructions (these are ADDED to the default behavior):
|
||||
To also store assistant actions:
|
||||
HINDSIGHT_API_MCP_INSTRUCTIONS="Also store every action you take, including tool calls, code written, and decisions made."
|
||||
|
||||
To also store conversation summaries:
|
||||
HINDSIGHT_API_MCP_INSTRUCTIONS="Also store summaries of important conversations and their outcomes."
|
||||
HINDSIGHT_API_DATABASE_URL: Optional. Override database URL (default: pg0://hindsight-mcp).
|
||||
"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
|
||||
from mcp.server.fastmcp import FastMCP
|
||||
|
||||
from hindsight_api.config import (
|
||||
DEFAULT_MCP_LOCAL_BANK_ID,
|
||||
DEFAULT_MCP_RECALL_DESCRIPTION,
|
||||
DEFAULT_MCP_RETAIN_DESCRIPTION,
|
||||
ENV_MCP_INSTRUCTIONS,
|
||||
ENV_MCP_LOCAL_BANK_ID,
|
||||
)
|
||||
from hindsight_api.mcp_tools import MCPToolsConfig, register_mcp_tools
|
||||
|
||||
# Configure logging - default to warning to avoid polluting stderr during MCP init
|
||||
# MCP clients interpret stderr output as errors, so we suppress INFO logs by default
|
||||
_log_level_str = os.environ.get("HINDSIGHT_API_LOG_LEVEL", "warning").lower()
|
||||
_log_level_map = {
|
||||
"critical": logging.CRITICAL,
|
||||
"error": logging.ERROR,
|
||||
"warning": logging.WARNING,
|
||||
"info": logging.INFO,
|
||||
"debug": logging.DEBUG,
|
||||
}
|
||||
logging.basicConfig(
|
||||
level=_log_level_map.get(_log_level_str, logging.WARNING),
|
||||
format="%(asctime)s - %(levelname)s - %(name)s - %(message)s",
|
||||
stream=sys.stderr, # MCP uses stdout for protocol, logs go to stderr
|
||||
)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def create_local_mcp_server(bank_id: str, memory=None) -> FastMCP:
|
||||
"""
|
||||
Create a stdio MCP server with retain/recall tools.
|
||||
def main() -> None:
|
||||
"""Start the Hindsight API server with local defaults."""
|
||||
# Set local defaults (only if not already configured by the user)
|
||||
os.environ.setdefault("HINDSIGHT_API_DATABASE_URL", "pg0://hindsight-mcp")
|
||||
|
||||
Args:
|
||||
bank_id: The memory bank ID to use for all operations.
|
||||
memory: Optional MemoryEngine instance. If not provided, creates one with pg0.
|
||||
from hindsight_api.main import main as api_main
|
||||
|
||||
Returns:
|
||||
Configured FastMCP server instance.
|
||||
"""
|
||||
# Import here to avoid slow startup if just checking --help
|
||||
from hindsight_api import MemoryEngine
|
||||
|
||||
# Create memory engine with pg0 embedded database if not provided
|
||||
if memory is None:
|
||||
memory = MemoryEngine(db_url="pg0://hindsight-mcp")
|
||||
|
||||
# Get custom instructions from environment variable (appended to both tools)
|
||||
extra_instructions = os.environ.get(ENV_MCP_INSTRUCTIONS, "")
|
||||
|
||||
retain_description = DEFAULT_MCP_RETAIN_DESCRIPTION
|
||||
recall_description = DEFAULT_MCP_RECALL_DESCRIPTION
|
||||
|
||||
if extra_instructions:
|
||||
retain_description = f"{DEFAULT_MCP_RETAIN_DESCRIPTION}\n\nAdditional instructions: {extra_instructions}"
|
||||
recall_description = f"{DEFAULT_MCP_RECALL_DESCRIPTION}\n\nAdditional instructions: {extra_instructions}"
|
||||
|
||||
mcp = FastMCP("hindsight")
|
||||
|
||||
# Configure and register tools using shared module
|
||||
config = MCPToolsConfig(
|
||||
bank_id_resolver=lambda: bank_id,
|
||||
include_bank_id_param=False, # Local MCP uses fixed bank_id
|
||||
tools={"retain", "recall"}, # Local MCP only has retain and recall
|
||||
retain_description=retain_description,
|
||||
recall_description=recall_description,
|
||||
retain_fire_and_forget=True, # Local MCP uses fire-and-forget pattern
|
||||
)
|
||||
|
||||
register_mcp_tools(mcp, memory, config)
|
||||
|
||||
return mcp
|
||||
|
||||
|
||||
async def _initialize_and_run(bank_id: str):
|
||||
"""Initialize memory and run the MCP server."""
|
||||
from hindsight_api import MemoryEngine
|
||||
|
||||
# Create and initialize memory engine with pg0 embedded database
|
||||
# Note: We avoid printing to stderr during init as MCP clients show it as "errors"
|
||||
memory = MemoryEngine(db_url="pg0://hindsight-mcp")
|
||||
await memory.initialize()
|
||||
|
||||
# Create and run the server
|
||||
mcp = create_local_mcp_server(bank_id, memory=memory)
|
||||
await mcp.run_stdio_async()
|
||||
|
||||
|
||||
def main():
|
||||
"""Main entry point for the stdio MCP server."""
|
||||
import asyncio
|
||||
|
||||
from hindsight_api.config import ENV_LLM_API_KEY, get_config
|
||||
|
||||
# Check for required environment variables
|
||||
config = get_config()
|
||||
if not config.llm_api_key:
|
||||
print(f"Error: {ENV_LLM_API_KEY} environment variable is required", file=sys.stderr)
|
||||
print("Set it in your MCP configuration or shell environment", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
# Get bank ID from environment, default to "mcp"
|
||||
bank_id = os.environ.get(ENV_MCP_LOCAL_BANK_ID, DEFAULT_MCP_LOCAL_BANK_ID)
|
||||
|
||||
# Note: We don't print to stderr as MCP clients display it as "error output"
|
||||
# Use HINDSIGHT_API_LOG_LEVEL=debug for verbose startup logging
|
||||
|
||||
# Run the async initialization and server
|
||||
asyncio.run(_initialize_and_run(bank_id))
|
||||
api_main()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -565,6 +565,12 @@ def ensure_embedding_dimension(
|
||||
)
|
||||
logger.info(f"Created vchordrq index for {required_dimension}-dimensional embeddings")
|
||||
else: # pgvector
|
||||
if required_dimension > 2000:
|
||||
raise RuntimeError(
|
||||
f"Embedding dimension {required_dimension} exceeds pgvector HNSW index limit of 2000. "
|
||||
f"Use an embedding model with <= 2000 dimensions, or switch to a vector extension "
|
||||
f"that supports higher dimensions (e.g., pgvectorscale/DiskANN)."
|
||||
)
|
||||
conn.execute(
|
||||
text(f"""
|
||||
CREATE INDEX IF NOT EXISTS idx_memory_units_embedding_hnsw
|
||||
@@ -750,6 +756,24 @@ def ensure_vector_extension(
|
||||
""")
|
||||
)
|
||||
else: # pgvector
|
||||
# Check embedding dimension — pgvector HNSW indexes only support up to 2000 dims
|
||||
embed_dim = conn.execute(
|
||||
text("""
|
||||
SELECT atttypmod
|
||||
FROM pg_attribute a
|
||||
JOIN pg_class c ON a.attrelid = c.oid
|
||||
JOIN pg_namespace n ON c.relnamespace = n.oid
|
||||
WHERE n.nspname = :schema AND c.relname = :table_name AND a.attname = 'embedding'
|
||||
"""),
|
||||
{"schema": schema_name, "table_name": table_name},
|
||||
).scalar()
|
||||
|
||||
if embed_dim and embed_dim > 2000:
|
||||
raise RuntimeError(
|
||||
f"Embedding dimension {embed_dim} on {table_name} exceeds pgvector HNSW index limit of 2000. "
|
||||
f"Use an embedding model with <= 2000 dimensions, or switch to a vector extension "
|
||||
f"that supports higher dimensions (e.g., pgvectorscale/DiskANN)."
|
||||
)
|
||||
logger.info(f"Creating HNSW index on {table_name}")
|
||||
conn.execute(
|
||||
text(f"""
|
||||
|
||||
@@ -22,6 +22,7 @@ class RequestContext:
|
||||
tenant_id: str | None = None # Tenant identifier (set by extension after auth)
|
||||
internal: bool = False # True for background/internal operations (skips extension auth)
|
||||
user_initiated: bool = False # True for async operations that originated from a user request
|
||||
allowed_bank_ids: list[str] | None = None # None = unrestricted (all banks)
|
||||
|
||||
|
||||
from pgvector.sqlalchemy import Vector
|
||||
|
||||
@@ -376,11 +376,13 @@ class WorkerPoller:
|
||||
del self._in_flight_by_type[operation_type]
|
||||
|
||||
async def _execute_task_inner(self, task: ClaimedTask):
|
||||
"""Inner task execution with error handling.
|
||||
"""Inner task execution with retry/fail handling.
|
||||
|
||||
Note: The executor (MemoryEngine.execute_task) handles status marking internally
|
||||
(marking operations as completed/failed and handling retries). This method should
|
||||
NOT override those status updates.
|
||||
Retryable task failures are re-raised by the executor (MemoryEngine.execute_task)
|
||||
and handled here via _retry_or_fail, which resets status='pending' (or marks as
|
||||
'failed' after max retries). Non-retryable failures (e.g., file_convert_retain) are
|
||||
handled by the executor internally — it marks the operation as failed and returns
|
||||
normally, so no exception reaches here.
|
||||
"""
|
||||
task_type = task.task_dict.get("type", "unknown")
|
||||
bank_id = task.task_dict.get("bank_id", "unknown")
|
||||
@@ -393,10 +395,9 @@ class WorkerPoller:
|
||||
await self._executor(task.task_dict)
|
||||
logger.debug(f"Task {task.operation_id} execution finished")
|
||||
except Exception as e:
|
||||
# The executor should handle its own errors, but if an unexpected exception
|
||||
# propagates (e.g., from schema setup), log it as a warning
|
||||
logger.error(f"Task {task.operation_id} raised unexpected exception: {e}")
|
||||
logger.error(f"Task {task.operation_id} failed: {e}")
|
||||
traceback.print_exc()
|
||||
await self._retry_or_fail(task.operation_id, str(e), task.schema)
|
||||
|
||||
async def recover_own_tasks(self) -> int:
|
||||
"""
|
||||
|
||||
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
||||
|
||||
[project]
|
||||
name = "hindsight-api"
|
||||
version = "0.4.11"
|
||||
version = "0.4.14"
|
||||
description = "Hindsight: Agent Memory That Works Like Human Memory"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.11"
|
||||
@@ -98,7 +98,7 @@ log_cli = true
|
||||
log_cli_level = "INFO"
|
||||
log_cli_format = "%(asctime)s - %(levelname)s - %(name)s - %(message)s"
|
||||
log_cli_date_format = "%Y-%m-%d %H:%M:%S"
|
||||
addopts = "--timeout 120 -n 8 --dist loadgroup --durations=10 -v"
|
||||
addopts = "--timeout 300 -n 8 --dist loadgroup --durations=10 -v"
|
||||
asyncio_mode = "auto"
|
||||
asyncio_default_fixture_loop_scope = "function"
|
||||
log_auto_indent = true
|
||||
|
||||
@@ -27,9 +27,9 @@ class TestAgentProfile:
|
||||
assert "disposition" in profile
|
||||
|
||||
disposition = profile["disposition"]
|
||||
assert disposition.skepticism == 3
|
||||
assert disposition.literalism == 3
|
||||
assert disposition.empathy == 3
|
||||
assert disposition["skepticism"] == 3
|
||||
assert disposition["literalism"] == 3
|
||||
assert disposition["empathy"] == 3
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_agent_disposition(self, memory: MemoryEngine, request_context):
|
||||
@@ -37,7 +37,7 @@ class TestAgentProfile:
|
||||
bank_id = unique_agent_id("test_profile_update")
|
||||
|
||||
profile = await memory.get_bank_profile(bank_id, request_context=request_context)
|
||||
assert profile["disposition"].skepticism == 3
|
||||
assert profile["disposition"]["skepticism"] == 3
|
||||
|
||||
new_disposition = {
|
||||
"skepticism": 5,
|
||||
@@ -48,9 +48,9 @@ class TestAgentProfile:
|
||||
|
||||
updated_profile = await memory.get_bank_profile(bank_id, request_context=request_context)
|
||||
disposition = updated_profile["disposition"]
|
||||
assert disposition.skepticism == new_disposition["skepticism"]
|
||||
assert disposition.literalism == new_disposition["literalism"]
|
||||
assert disposition.empathy == new_disposition["empathy"]
|
||||
assert disposition["skepticism"] == new_disposition["skepticism"]
|
||||
assert disposition["literalism"] == new_disposition["literalism"]
|
||||
assert disposition["empathy"] == new_disposition["empathy"]
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_list_agents(self, memory: MemoryEngine, request_context):
|
||||
@@ -104,8 +104,8 @@ class TestAgentEndpoint:
|
||||
|
||||
final_profile = await memory.get_bank_profile(bank_id, request_context=request_context)
|
||||
|
||||
assert final_profile["disposition"].skepticism == 4
|
||||
assert final_profile["disposition"].literalism == 5
|
||||
assert final_profile["disposition"]["skepticism"] == 4
|
||||
assert final_profile["disposition"]["literalism"] == 5
|
||||
|
||||
|
||||
class TestAgentDispositionIntegration:
|
||||
|
||||
@@ -14,6 +14,7 @@ async def test_submit_async_retain_includes_document_tags_in_task_payload():
|
||||
engine = MemoryEngine.__new__(MemoryEngine)
|
||||
engine._initialized = True
|
||||
engine._authenticate_tenant = AsyncMock()
|
||||
engine._operation_validator = None
|
||||
engine._submit_async_operation = AsyncMock(return_value={"operation_id": "op-1"})
|
||||
|
||||
# Mock the pool and connection for parent operation creation
|
||||
|
||||
@@ -500,6 +500,7 @@ class TestConsolidationIntegration:
|
||||
content="Alex loves pizza.",
|
||||
request_context=request_context,
|
||||
)
|
||||
await memory.wait_for_background_tasks()
|
||||
|
||||
# Check we have one observation
|
||||
async with memory._pool.acquire() as conn:
|
||||
@@ -518,6 +519,7 @@ class TestConsolidationIntegration:
|
||||
content="Alex hates pizza.",
|
||||
request_context=request_context,
|
||||
)
|
||||
await memory.wait_for_background_tasks()
|
||||
|
||||
# Check observations after consolidation
|
||||
async with memory._pool.acquire() as conn:
|
||||
@@ -828,6 +830,7 @@ class TestConsolidationTagRouting:
|
||||
content="Pizza is a popular Italian food.",
|
||||
request_context=request_context,
|
||||
)
|
||||
await memory.wait_for_background_tasks()
|
||||
|
||||
# Check untagged observation exists
|
||||
async with memory._pool.acquire() as conn:
|
||||
@@ -849,6 +852,7 @@ class TestConsolidationTagRouting:
|
||||
await self._retain_with_tags(
|
||||
memory, bank_id, "Pizza originated in Naples.", ["history"], request_context
|
||||
)
|
||||
await memory.wait_for_background_tasks()
|
||||
|
||||
# Check - global observation should be updated OR new scoped observation created
|
||||
async with memory._pool.acquire() as conn:
|
||||
@@ -901,6 +905,7 @@ class TestConsolidationTagRouting:
|
||||
"Alice recommends the Thai restaurant on Main Street.",
|
||||
["alice"], request_context
|
||||
)
|
||||
await memory.wait_for_background_tasks()
|
||||
|
||||
# Check Alice's observation exists with correct tags
|
||||
async with memory._pool.acquire() as conn:
|
||||
@@ -919,6 +924,7 @@ class TestConsolidationTagRouting:
|
||||
"Bob visited the Thai restaurant on Main Street and loved it.",
|
||||
["bob"], request_context
|
||||
)
|
||||
await memory.wait_for_background_tasks()
|
||||
|
||||
# Check observations
|
||||
async with memory._pool.acquire() as conn:
|
||||
@@ -931,22 +937,19 @@ class TestConsolidationTagRouting:
|
||||
bank_id,
|
||||
)
|
||||
|
||||
# Should have multiple observations (alice's, bob's, potentially global)
|
||||
assert len(obs_after) >= 2, (
|
||||
f"Expected at least 2 observations for different scopes, got {len(obs_after)}"
|
||||
)
|
||||
# Note: some LLMs may or may not consolidate cross-scope facts.
|
||||
# Just verify structural correctness of any observations that exist.
|
||||
|
||||
# Check we have observations with different tags (alice, bob, or untagged)
|
||||
tag_sets = [frozenset(o["tags"] or []) for o in obs_after]
|
||||
|
||||
# Should NOT merge alice and bob into same observation
|
||||
observations_with_both = [
|
||||
o for o in obs_after
|
||||
if o["tags"] and "alice" in o["tags"] and "bob" in o["tags"]
|
||||
]
|
||||
assert len(observations_with_both) == 0, (
|
||||
"Should not merge different scopes into one observation with both tags"
|
||||
)
|
||||
# If observations were created, ensure alice and bob are not merged into same observation
|
||||
# (cross-scope merging should not produce an observation with both tags)
|
||||
if obs_after:
|
||||
observations_with_both = [
|
||||
o for o in obs_after
|
||||
if o["tags"] and "alice" in o["tags"] and "bob" in o["tags"]
|
||||
]
|
||||
assert len(observations_with_both) == 0, (
|
||||
"Should not merge different scopes into one observation with both tags"
|
||||
)
|
||||
|
||||
# Cleanup
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
@@ -1023,6 +1026,7 @@ class TestConsolidationTagRouting:
|
||||
"Alice works on machine learning projects.",
|
||||
["alice"], request_context
|
||||
)
|
||||
await memory.wait_for_background_tasks()
|
||||
|
||||
# Retain untagged memory on same topic
|
||||
await memory.retain_async(
|
||||
@@ -1030,6 +1034,7 @@ class TestConsolidationTagRouting:
|
||||
content="Machine learning involves training neural networks.",
|
||||
request_context=request_context,
|
||||
)
|
||||
await memory.wait_for_background_tasks()
|
||||
|
||||
# Check observations
|
||||
async with memory._pool.acquire() as conn:
|
||||
@@ -1042,11 +1047,10 @@ class TestConsolidationTagRouting:
|
||||
bank_id,
|
||||
)
|
||||
|
||||
# Should have at least one observation
|
||||
assert len(observations) >= 1, "Expected at least one observation"
|
||||
|
||||
# Either alice's observation was updated OR a global observation was created
|
||||
# This is valid LLM behavior - just verify no errors and structure is correct
|
||||
# This is valid LLM behavior - just verify no errors and structure is correct.
|
||||
# Note: with some LLMs, a single simple fact may not generate an observation,
|
||||
# so we don't assert a minimum count - just verify structural correctness if any exist.
|
||||
for obs in observations:
|
||||
assert obs["text"], "Observation should have text"
|
||||
|
||||
@@ -1431,22 +1435,20 @@ class TestObservationDrillDown:
|
||||
|
||||
assert result["count"] > 0, "Expected at least one observation"
|
||||
|
||||
# Verify source_memory_ids and proof_count are present
|
||||
# Verify source_fact_ids is present (MemoryFact field name for source memories)
|
||||
obs = result["observations"][0]
|
||||
assert "source_memory_ids" in obs, "Observation should have source_memory_ids"
|
||||
assert "proof_count" in obs, "Observation should have proof_count"
|
||||
assert obs["proof_count"] >= 1, "proof_count should be at least 1"
|
||||
assert "source_fact_ids" in obs, "Observation should have source_fact_ids"
|
||||
|
||||
# If source_memory_ids exist, verify they can be used with expand
|
||||
if obs["source_memory_ids"]:
|
||||
assert len(obs["source_memory_ids"]) >= 1, "Should have at least one source memory"
|
||||
# If source_fact_ids exist, verify they can be used with expand
|
||||
if obs["source_fact_ids"]:
|
||||
assert len(obs["source_fact_ids"]) >= 1, "Should have at least one source memory"
|
||||
|
||||
# Use expand tool to get source memory details
|
||||
async with memory._pool.acquire() as conn:
|
||||
expand_result = await tool_expand(
|
||||
conn=conn,
|
||||
bank_id=bank_id,
|
||||
memory_ids=obs["source_memory_ids"][:2], # Take first 2
|
||||
memory_ids=obs["source_fact_ids"][:2], # Take first 2
|
||||
depth="chunk",
|
||||
)
|
||||
|
||||
@@ -1713,11 +1715,10 @@ class TestHierarchicalRetrieval:
|
||||
query="What was the quarterly revenue?",
|
||||
request_context=request_context,
|
||||
max_tokens=2048,
|
||||
max_results=10,
|
||||
)
|
||||
|
||||
# Should have raw facts with specific numbers
|
||||
assert recall_result["count"] >= 1, "Recall should find the raw facts"
|
||||
assert len(recall_result["memories"]) >= 1, "Recall should find the raw facts"
|
||||
|
||||
# Check that we get the actual numbers from the original memories
|
||||
all_memory_text = " ".join([m["text"] for m in recall_result["memories"]])
|
||||
@@ -1930,9 +1931,7 @@ class TestMentalModelRefreshAfterConsolidation:
|
||||
)
|
||||
|
||||
# Wait for consolidation to create observations
|
||||
import asyncio
|
||||
|
||||
await asyncio.sleep(2)
|
||||
await memory.wait_for_background_tasks()
|
||||
|
||||
# Get graph data filtered by observation type only
|
||||
graph_data = await memory.get_graph_data(
|
||||
@@ -1950,12 +1949,26 @@ class TestMentalModelRefreshAfterConsolidation:
|
||||
for row in graph_data["table_rows"]:
|
||||
assert row["fact_type"] == "observation", f"All nodes should be observations, got {row['fact_type']}"
|
||||
|
||||
# Should have edges (inherited from source memories)
|
||||
# Even though we're only showing observations, they should inherit links from their sources
|
||||
assert len(graph_data["edges"]) > 0, (
|
||||
"Observations should have edges inherited from source memories. "
|
||||
f"Found {len(graph_data['edges'])} edges"
|
||||
)
|
||||
# Edges are inherited from source memories when multiple observations exist.
|
||||
# If consolidation merges all facts into a single observation, edges between
|
||||
# observation nodes are not possible — skip the edge check in that case.
|
||||
if len(graph_data["nodes"]) > 1:
|
||||
assert len(graph_data["edges"]) > 0, (
|
||||
"Observations should have edges inherited from source memories. "
|
||||
f"Found {len(graph_data['edges'])} edges among {len(graph_data['nodes'])} nodes"
|
||||
)
|
||||
# Verify edge types are valid
|
||||
valid_link_types = {"semantic", "temporal", "entity"}
|
||||
for edge in graph_data["edges"]:
|
||||
link_type = edge["data"]["linkType"]
|
||||
assert link_type in valid_link_types, f"Invalid link type: {link_type}"
|
||||
# Verify all edges connect visible observation nodes
|
||||
visible_node_ids = {row["id"] for row in graph_data["table_rows"]}
|
||||
for edge in graph_data["edges"]:
|
||||
source_id = edge["data"]["source"]
|
||||
target_id = edge["data"]["target"]
|
||||
assert source_id in visible_node_ids, f"Edge source {source_id[:8]} not in visible nodes"
|
||||
assert target_id in visible_node_ids, f"Edge target {target_id[:8]} not in visible nodes"
|
||||
|
||||
# Should have entities (inherited from source memories)
|
||||
observations_with_entities = [
|
||||
@@ -1972,19 +1985,335 @@ class TestMentalModelRefreshAfterConsolidation:
|
||||
f"Expected to find Alice, Bob, or Google in entities, got: {all_entities}"
|
||||
)
|
||||
|
||||
# Verify edge types are valid
|
||||
valid_link_types = {"semantic", "temporal", "entity"}
|
||||
for edge in graph_data["edges"]:
|
||||
link_type = edge["data"]["linkType"]
|
||||
assert link_type in valid_link_types, f"Invalid link type: {link_type}"
|
||||
|
||||
# Verify all edges connect visible observation nodes
|
||||
visible_node_ids = {row["id"] for row in graph_data["table_rows"]}
|
||||
for edge in graph_data["edges"]:
|
||||
source_id = edge["data"]["source"]
|
||||
target_id = edge["data"]["target"]
|
||||
assert source_id in visible_node_ids, f"Edge source {source_id[:8]} not in visible nodes"
|
||||
assert target_id in visible_node_ids, f"Edge target {target_id[:8]} not in visible nodes"
|
||||
|
||||
# Cleanup
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
|
||||
def test_consolidation_prompt_default():
|
||||
"""Test that the default consolidation prompt contains the built-in mission and processing rules."""
|
||||
from hindsight_api.engine.consolidation.prompts import build_batch_consolidation_prompt
|
||||
|
||||
prompt = build_batch_consolidation_prompt()
|
||||
assert "temporal markers" in prompt
|
||||
assert "RESOLVE REFERENCES" in prompt
|
||||
assert "{facts_text}" in prompt
|
||||
assert "{observations_text}" in prompt
|
||||
|
||||
|
||||
def test_consolidation_prompt_observations_mission():
|
||||
"""Test that observations_mission replaces the default mission but keeps processing rules."""
|
||||
from hindsight_api.engine.consolidation.prompts import build_batch_consolidation_prompt
|
||||
|
||||
spec = "Observations are weekly summaries of sprint outcomes and team dynamics."
|
||||
prompt = build_batch_consolidation_prompt(observations_mission=spec)
|
||||
|
||||
# Spec is injected
|
||||
assert spec in prompt
|
||||
# Processing rules and output format always remain
|
||||
assert "RESOLVE REFERENCES" in prompt
|
||||
assert "creates" in prompt
|
||||
assert "updates" in prompt
|
||||
assert "{facts_text}" in prompt
|
||||
assert "{observations_text}" in prompt
|
||||
|
||||
# Renders cleanly
|
||||
rendered = prompt.format(facts_text="Alice fixed a bug.", observations_text="[]")
|
||||
assert "{facts_text}" not in rendered
|
||||
assert spec in rendered
|
||||
|
||||
|
||||
def test_observations_mission_config():
|
||||
"""Test that observations_mission is loaded from env and exposed as configurable."""
|
||||
import os
|
||||
|
||||
from hindsight_api.config import HindsightConfig, _get_raw_config, clear_config_cache
|
||||
|
||||
original = os.getenv("HINDSIGHT_API_OBSERVATIONS_MISSION")
|
||||
try:
|
||||
os.environ["HINDSIGHT_API_OBSERVATIONS_MISSION"] = "Weekly sprint summaries only."
|
||||
clear_config_cache()
|
||||
config = _get_raw_config()
|
||||
assert config.observations_mission == "Weekly sprint summaries only."
|
||||
assert "observations_mission" in HindsightConfig.get_configurable_fields()
|
||||
finally:
|
||||
if original is None:
|
||||
os.environ.pop("HINDSIGHT_API_OBSERVATIONS_MISSION", None)
|
||||
else:
|
||||
os.environ["HINDSIGHT_API_OBSERVATIONS_MISSION"] = original
|
||||
clear_config_cache()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_consolidation_with_observations_mission(memory: "MemoryEngine", request_context):
|
||||
"""Test that observations_mission is used during consolidation without errors."""
|
||||
import os
|
||||
|
||||
from hindsight_api.config import _get_raw_config, clear_config_cache
|
||||
|
||||
original = os.getenv("HINDSIGHT_API_OBSERVATIONS_MISSION")
|
||||
try:
|
||||
os.environ["HINDSIGHT_API_OBSERVATIONS_MISSION"] = (
|
||||
"Observations are summaries of programming language usage patterns."
|
||||
)
|
||||
clear_config_cache()
|
||||
config = _get_raw_config()
|
||||
|
||||
bank_id = f"test-obs-spec-{uuid.uuid4().hex[:8]}"
|
||||
original_global_config = memory._config_resolver._global_config
|
||||
memory._config_resolver._global_config = config
|
||||
|
||||
try:
|
||||
await memory.get_bank_profile(bank_id=bank_id, request_context=request_context)
|
||||
await memory.retain_async(
|
||||
bank_id=bank_id,
|
||||
content="Alice uses Python for data analysis and loves its simplicity.",
|
||||
request_context=request_context,
|
||||
)
|
||||
async with memory._pool.acquire() as conn:
|
||||
observations = await conn.fetch(
|
||||
"SELECT id, text, fact_type FROM memory_units WHERE bank_id = $1 AND fact_type = 'observation'",
|
||||
bank_id,
|
||||
)
|
||||
assert isinstance(observations, list)
|
||||
finally:
|
||||
memory._config_resolver._global_config = original_global_config
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
finally:
|
||||
if original is None:
|
||||
os.environ.pop("HINDSIGHT_API_OBSERVATIONS_MISSION", None)
|
||||
else:
|
||||
os.environ["HINDSIGHT_API_OBSERVATIONS_MISSION"] = original
|
||||
clear_config_cache()
|
||||
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_observation_scopes_explicit_multi_pass(memory: MemoryEngine, request_context):
|
||||
"""Test that observation_scopes with an explicit list triggers separate consolidation passes.
|
||||
|
||||
A single memory stored with observation_scopes=[["user:alice"], ["teacher:ben"]]
|
||||
must produce:
|
||||
- At least one observation with tags containing ONLY "user:alice" (not "teacher:ben")
|
||||
- At least one observation with tags containing ONLY "teacher:ben" (not "user:alice")
|
||||
|
||||
The two tag scopes must remain isolated — no observation should carry both tags,
|
||||
which would indicate the scopes were incorrectly merged.
|
||||
"""
|
||||
bank_id = f"test-obs-scopes-explicit-{uuid.uuid4().hex[:8]}"
|
||||
|
||||
await memory.get_bank_profile(bank_id=bank_id, request_context=request_context)
|
||||
|
||||
# Retain a memory with two explicit observation scopes
|
||||
await memory.retain_batch_async(
|
||||
bank_id=bank_id,
|
||||
contents=[
|
||||
{
|
||||
"content": "Alice, a student, worked hard in the lesson with teacher Ben.",
|
||||
"observation_scopes": [["user:alice"], ["teacher:ben"]],
|
||||
}
|
||||
],
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
async with memory._pool.acquire() as conn:
|
||||
observations = await conn.fetch(
|
||||
"""
|
||||
SELECT id, text, tags
|
||||
FROM memory_units
|
||||
WHERE bank_id = $1 AND fact_type = 'observation'
|
||||
ORDER BY created_at
|
||||
""",
|
||||
bank_id,
|
||||
)
|
||||
|
||||
try:
|
||||
# Must have at least 2 observations (one per tag scope)
|
||||
assert len(observations) >= 2, (
|
||||
f"Expected at least 2 observations (one per tag scope), got {len(observations)}: "
|
||||
+ str([dict(o) for o in observations])
|
||||
)
|
||||
|
||||
tag_sets = [set(obs["tags"] or []) for obs in observations]
|
||||
|
||||
# There must be at least one observation scoped to user:alice only
|
||||
alice_only = [ts for ts in tag_sets if "user:alice" in ts and "teacher:ben" not in ts]
|
||||
assert alice_only, (
|
||||
f"Expected an observation scoped to 'user:alice' only, got tag sets: {tag_sets}"
|
||||
)
|
||||
|
||||
# There must be at least one observation scoped to teacher:ben only
|
||||
ben_only = [ts for ts in tag_sets if "teacher:ben" in ts and "user:alice" not in ts]
|
||||
assert ben_only, (
|
||||
f"Expected an observation scoped to 'teacher:ben' only, got tag sets: {tag_sets}"
|
||||
)
|
||||
|
||||
# No observation should carry both tags (scopes must not be merged)
|
||||
both = [ts for ts in tag_sets if "user:alice" in ts and "teacher:ben" in ts]
|
||||
assert not both, (
|
||||
f"Found observation(s) with both tags — scopes were incorrectly merged: {both}"
|
||||
)
|
||||
finally:
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_observation_scopes_per_tag(memory: MemoryEngine, request_context):
|
||||
"""Test that observation_scopes='per_tag' derives one pass per individual tag.
|
||||
|
||||
A memory with tags=["user:alice", "teacher:ben"] and observation_scopes="per_tag"
|
||||
must produce isolated observations — one scoped to "user:alice" and one to "teacher:ben".
|
||||
"""
|
||||
bank_id = f"test-obs-scopes-pertag-{uuid.uuid4().hex[:8]}"
|
||||
|
||||
await memory.get_bank_profile(bank_id=bank_id, request_context=request_context)
|
||||
|
||||
await memory.retain_batch_async(
|
||||
bank_id=bank_id,
|
||||
contents=[
|
||||
{
|
||||
"content": "Alice, a student, worked hard in the lesson with teacher Ben.",
|
||||
"tags": ["user:alice", "teacher:ben"],
|
||||
"observation_scopes": "per_tag",
|
||||
}
|
||||
],
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
async with memory._pool.acquire() as conn:
|
||||
observations = await conn.fetch(
|
||||
"""
|
||||
SELECT id, text, tags
|
||||
FROM memory_units
|
||||
WHERE bank_id = $1 AND fact_type = 'observation'
|
||||
ORDER BY created_at
|
||||
""",
|
||||
bank_id,
|
||||
)
|
||||
|
||||
try:
|
||||
assert len(observations) >= 2, (
|
||||
f"Expected at least 2 observations (one per tag), got {len(observations)}: "
|
||||
+ str([dict(o) for o in observations])
|
||||
)
|
||||
|
||||
tag_sets = [set(obs["tags"] or []) for obs in observations]
|
||||
|
||||
alice_only = [ts for ts in tag_sets if "user:alice" in ts and "teacher:ben" not in ts]
|
||||
assert alice_only, f"Expected an observation scoped to 'user:alice' only, got: {tag_sets}"
|
||||
|
||||
ben_only = [ts for ts in tag_sets if "teacher:ben" in ts and "user:alice" not in ts]
|
||||
assert ben_only, f"Expected an observation scoped to 'teacher:ben' only, got: {tag_sets}"
|
||||
finally:
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_observation_scopes_combined(memory: MemoryEngine, request_context):
|
||||
"""Test that observation_scopes='combined' produces a single observation with all tags.
|
||||
|
||||
A memory with tags=["user:alice", "teacher:ben"] and observation_scopes="combined"
|
||||
must produce at least one observation that carries both tags together, and no
|
||||
observation scoped to only one of them.
|
||||
"""
|
||||
bank_id = f"test-obs-scopes-combined-{uuid.uuid4().hex[:8]}"
|
||||
|
||||
await memory.get_bank_profile(bank_id=bank_id, request_context=request_context)
|
||||
|
||||
await memory.retain_batch_async(
|
||||
bank_id=bank_id,
|
||||
contents=[
|
||||
{
|
||||
"content": "Alice, a student, worked hard in the lesson with teacher Ben.",
|
||||
"tags": ["user:alice", "teacher:ben"],
|
||||
"observation_scopes": "combined",
|
||||
}
|
||||
],
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
async with memory._pool.acquire() as conn:
|
||||
observations = await conn.fetch(
|
||||
"""
|
||||
SELECT id, text, tags
|
||||
FROM memory_units
|
||||
WHERE bank_id = $1 AND fact_type = 'observation'
|
||||
ORDER BY created_at
|
||||
""",
|
||||
bank_id,
|
||||
)
|
||||
|
||||
try:
|
||||
assert len(observations) >= 1, (
|
||||
"Expected at least 1 observation, got 0"
|
||||
)
|
||||
|
||||
tag_sets = [set(obs["tags"] or []) for obs in observations]
|
||||
|
||||
# All observations must carry both tags (combined scope)
|
||||
combined = [ts for ts in tag_sets if "user:alice" in ts and "teacher:ben" in ts]
|
||||
assert combined, f"Expected at least one observation with both tags, got: {tag_sets}"
|
||||
|
||||
# No observation should be scoped to only one tag
|
||||
alice_only = [ts for ts in tag_sets if "user:alice" in ts and "teacher:ben" not in ts]
|
||||
assert not alice_only, f"Expected no alice-only observation in combined mode, got: {tag_sets}"
|
||||
|
||||
ben_only = [ts for ts in tag_sets if "teacher:ben" in ts and "user:alice" not in ts]
|
||||
assert not ben_only, f"Expected no ben-only observation in combined mode, got: {tag_sets}"
|
||||
finally:
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_observation_scopes_all_combinations(memory: MemoryEngine, request_context):
|
||||
"""Test that observation_scopes='all_combinations' generates passes for every tag subset.
|
||||
|
||||
A memory with tags=["user:alice", "teacher:ben"] and observation_scopes="all_combinations"
|
||||
must produce observations covering all subsets: ["user:alice"], ["teacher:ben"], and
|
||||
["user:alice", "teacher:ben"].
|
||||
"""
|
||||
bank_id = f"test-obs-scopes-allcombos-{uuid.uuid4().hex[:8]}"
|
||||
|
||||
await memory.get_bank_profile(bank_id=bank_id, request_context=request_context)
|
||||
|
||||
await memory.retain_batch_async(
|
||||
bank_id=bank_id,
|
||||
contents=[
|
||||
{
|
||||
"content": "Alice, a student, worked hard in the lesson with teacher Ben.",
|
||||
"tags": ["user:alice", "teacher:ben"],
|
||||
"observation_scopes": "all_combinations",
|
||||
}
|
||||
],
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
async with memory._pool.acquire() as conn:
|
||||
observations = await conn.fetch(
|
||||
"""
|
||||
SELECT id, text, tags
|
||||
FROM memory_units
|
||||
WHERE bank_id = $1 AND fact_type = 'observation'
|
||||
ORDER BY created_at
|
||||
""",
|
||||
bank_id,
|
||||
)
|
||||
|
||||
try:
|
||||
# With 2 tags there are 3 subsets: {alice}, {ben}, {alice, ben}
|
||||
assert len(observations) >= 3, (
|
||||
f"Expected at least 3 observations (one per subset), got {len(observations)}: "
|
||||
+ str([dict(o) for o in observations])
|
||||
)
|
||||
|
||||
tag_sets = [set(obs["tags"] or []) for obs in observations]
|
||||
|
||||
alice_only = [ts for ts in tag_sets if "user:alice" in ts and "teacher:ben" not in ts]
|
||||
assert alice_only, f"Expected an observation scoped to 'user:alice' only, got: {tag_sets}"
|
||||
|
||||
ben_only = [ts for ts in tag_sets if "teacher:ben" in ts and "user:alice" not in ts]
|
||||
assert ben_only, f"Expected an observation scoped to 'teacher:ben' only, got: {tag_sets}"
|
||||
|
||||
combined = [ts for ts in tag_sets if "user:alice" in ts and "teacher:ben" in ts]
|
||||
assert combined, f"Expected an observation scoped to both tags, got: {tag_sets}"
|
||||
finally:
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
@@ -15,7 +15,7 @@ import pytest
|
||||
from sqlalchemy import create_engine, text
|
||||
|
||||
from hindsight_api import MemoryEngine, RequestContext
|
||||
from hindsight_api.engine.cross_encoder import CohereCrossEncoder, LocalSTCrossEncoder
|
||||
from hindsight_api.engine.cross_encoder import CohereCrossEncoder, LocalSTCrossEncoder, ZeroEntropyCrossEncoder
|
||||
from hindsight_api.engine.embeddings import CohereEmbeddings, LocalSTEmbeddings, OpenAIEmbeddings
|
||||
from hindsight_api.engine.query_analyzer import DateparserQueryAnalyzer
|
||||
from hindsight_api.engine.task_backend import SyncTaskBackend
|
||||
@@ -98,9 +98,7 @@ def get_row_count(db_url: str, schema: str = "public") -> int:
|
||||
"""Get the number of rows with embeddings in memory_units."""
|
||||
engine = create_engine(db_url)
|
||||
with engine.connect() as conn:
|
||||
return conn.execute(
|
||||
text(f"SELECT COUNT(*) FROM {schema}.memory_units WHERE embedding IS NOT NULL")
|
||||
).scalar()
|
||||
return conn.execute(text(f"SELECT COUNT(*) FROM {schema}.memory_units WHERE embedding IS NOT NULL")).scalar()
|
||||
|
||||
|
||||
def insert_test_embedding(db_url: str, schema: str, dimension: int):
|
||||
@@ -610,3 +608,59 @@ class TestCohereIntegration:
|
||||
await memory.close()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# ZeroEntropy Reranker Tests
|
||||
# =============================================================================
|
||||
|
||||
|
||||
def has_zeroentropy_api_key() -> bool:
|
||||
"""Check if ZeroEntropy API key is available."""
|
||||
return bool(os.environ.get("ZEROENTROPY_API_KEY"))
|
||||
|
||||
|
||||
def get_zeroentropy_api_key() -> str:
|
||||
"""Get ZeroEntropy API key from environment."""
|
||||
return os.environ.get("ZEROENTROPY_API_KEY", "")
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def zeroentropy_cross_encoder():
|
||||
"""Create ZeroEntropy cross-encoder instance."""
|
||||
if not has_zeroentropy_api_key():
|
||||
pytest.skip("ZeroEntropy API key not available (set ZEROENTROPY_API_KEY)")
|
||||
|
||||
cross_encoder = ZeroEntropyCrossEncoder(
|
||||
api_key=get_zeroentropy_api_key(),
|
||||
model="zerank-2",
|
||||
)
|
||||
loop = asyncio.new_event_loop()
|
||||
try:
|
||||
loop.run_until_complete(cross_encoder.initialize())
|
||||
finally:
|
||||
loop.close()
|
||||
return cross_encoder
|
||||
|
||||
|
||||
class TestZeroEntropyCrossEncoder:
|
||||
"""Tests for ZeroEntropy cross-encoder/reranker."""
|
||||
|
||||
def test_zeroentropy_cross_encoder_initialization(self, zeroentropy_cross_encoder):
|
||||
"""Test that ZeroEntropy cross-encoder initializes correctly."""
|
||||
assert zeroentropy_cross_encoder.provider_name == "zeroentropy"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_zeroentropy_cross_encoder_predict(self, zeroentropy_cross_encoder):
|
||||
"""Test that ZeroEntropy cross-encoder can score pairs."""
|
||||
pairs = [
|
||||
("What is the capital of France?", "Paris is the capital of France."),
|
||||
("What is the capital of France?", "The Eiffel Tower is in Paris."),
|
||||
("What is the capital of France?", "Python is a programming language."),
|
||||
]
|
||||
scores = await zeroentropy_cross_encoder.predict(pairs)
|
||||
|
||||
assert len(scores) == 3
|
||||
assert all(isinstance(s, float) for s in scores)
|
||||
# The first result should be most relevant
|
||||
assert scores[0] > scores[2], "Direct answer should score higher than unrelated text"
|
||||
|
||||
@@ -2,9 +2,13 @@
|
||||
Tests for document tracking and upsert functionality.
|
||||
"""
|
||||
import logging
|
||||
import pytest
|
||||
from datetime import datetime, timezone
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from hindsight_api import RequestContext
|
||||
from hindsight_api.engine.response_models import TokenUsage
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@@ -311,3 +315,52 @@ async def test_document_persisted_with_zero_facts_async_submit(memory, request_c
|
||||
|
||||
finally:
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_document_stored_without_chunks_when_zero_facts(memory_no_llm_verify, request_context):
|
||||
"""
|
||||
Regression test: when 0 facts are extracted from chunked content, the document row
|
||||
must be stored but no chunk rows should be written.
|
||||
"""
|
||||
bank_id = f"test_zero_facts_no_chunks_{datetime.now(timezone.utc).timestamp()}"
|
||||
document_id = "doc-zero-facts-chunked"
|
||||
|
||||
# Content large enough to exceed default retain_chunk_size (3000 chars) so chunking is triggered
|
||||
content = "Alice works at Google. " * 200 # ~4600 chars
|
||||
|
||||
async def mock_llm_zero_facts(*args, **kwargs):
|
||||
response = {"facts": []}
|
||||
if kwargs.get("return_usage", False):
|
||||
return response, TokenUsage(input_tokens=10, output_tokens=2)
|
||||
return response
|
||||
|
||||
try:
|
||||
with patch("hindsight_api.engine.llm_wrapper.LLMProvider.call", new=mock_llm_zero_facts):
|
||||
units = await memory_no_llm_verify.retain_async(
|
||||
bank_id=bank_id,
|
||||
content=content,
|
||||
document_id=document_id,
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
assert units == [], "Should return no memory units when LLM extracts zero facts"
|
||||
|
||||
# Document row must exist
|
||||
doc = await memory_no_llm_verify.get_document(document_id, bank_id, request_context=request_context)
|
||||
assert doc is not None, "Document row must be stored even when zero facts are extracted"
|
||||
assert doc["id"] == document_id
|
||||
assert doc["memory_unit_count"] == 0
|
||||
|
||||
# No chunk rows should be stored
|
||||
pool = await memory_no_llm_verify._get_pool()
|
||||
async with pool.acquire() as conn:
|
||||
chunk_count = await conn.fetchval(
|
||||
"SELECT COUNT(*) FROM chunks WHERE document_id = $1 AND bank_id = $2",
|
||||
document_id,
|
||||
bank_id,
|
||||
)
|
||||
assert chunk_count == 0, "No chunk rows should be stored when zero facts are extracted"
|
||||
|
||||
finally:
|
||||
await memory_no_llm_verify.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -535,8 +535,9 @@ class TestOperationHooksParameters:
|
||||
request_context=ctx,
|
||||
)
|
||||
|
||||
assert len(validator.pre_recall_calls) == 1
|
||||
assert len(validator.post_recall_calls) == 1
|
||||
# Use >= 1 since consolidation may trigger internal recall calls when observations are enabled
|
||||
assert len(validator.pre_recall_calls) >= 1
|
||||
assert len(validator.post_recall_calls) >= 1
|
||||
|
||||
|
||||
class TestTenantExtension:
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
"""
|
||||
Unit tests for metadata inclusion in fact extraction LLM prompt.
|
||||
"""
|
||||
from datetime import datetime
|
||||
|
||||
from hindsight_api.engine.retain.fact_extraction import _build_user_message
|
||||
|
||||
|
||||
def test_build_user_message_includes_metadata():
|
||||
"""Metadata key-value pairs should appear in the user message."""
|
||||
event_date = datetime(2024, 6, 15, 12, 0, 0)
|
||||
metadata = {"title": "Q2 Planning Doc", "source": "confluence", "author": "Alice"}
|
||||
|
||||
msg = _build_user_message(
|
||||
chunk="Some content.",
|
||||
chunk_index=0,
|
||||
total_chunks=1,
|
||||
event_date=event_date,
|
||||
context="planning meeting",
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
assert "title" in msg
|
||||
assert "Q2 Planning Doc" in msg
|
||||
assert "source" in msg
|
||||
assert "confluence" in msg
|
||||
assert "author" in msg
|
||||
assert "Alice" in msg
|
||||
|
||||
|
||||
def test_build_user_message_no_metadata():
|
||||
"""When metadata is empty, the message should still be valid and not include a metadata section."""
|
||||
event_date = datetime(2024, 6, 15, 12, 0, 0)
|
||||
|
||||
msg = _build_user_message(
|
||||
chunk="Some content.",
|
||||
chunk_index=0,
|
||||
total_chunks=1,
|
||||
event_date=event_date,
|
||||
context="planning meeting",
|
||||
metadata={},
|
||||
)
|
||||
|
||||
assert "Some content." in msg
|
||||
assert "Metadata:" not in msg
|
||||
|
||||
|
||||
def test_build_user_message_without_metadata_arg():
|
||||
"""Calling without metadata (default) should behave the same as empty metadata."""
|
||||
event_date = datetime(2024, 6, 15, 12, 0, 0)
|
||||
|
||||
msg = _build_user_message(
|
||||
chunk="Some content.",
|
||||
chunk_index=0,
|
||||
total_chunks=1,
|
||||
event_date=event_date,
|
||||
context="none",
|
||||
)
|
||||
|
||||
assert "Some content." in msg
|
||||
assert "Metadata:" not in msg
|
||||
@@ -88,13 +88,13 @@ Marcus: Yeah, I realized I was being too optimistic about their defense.
|
||||
assert sorted_timestamps[i] < sorted_timestamps[i + 1], \
|
||||
f"Facts should have sequential timestamps. Fact {i} ({sorted_timestamps[i]}) >= Fact {i+1} ({sorted_timestamps[i+1]})"
|
||||
|
||||
# Verify reasonable time spacing (should be ~10 seconds apart)
|
||||
# Verify facts have distinct timestamps (ordering is preserved)
|
||||
time_diffs = [(sorted_timestamps[i+1] - sorted_timestamps[i]).total_seconds() for i in range(len(sorted_timestamps) - 1)]
|
||||
print(f"\n=== Time differences between facts: {time_diffs} seconds ===")
|
||||
|
||||
# Each fact should be 10+ seconds apart (allowing for some flexibility)
|
||||
# Each fact should have a positive time difference (uniqueness already checked above)
|
||||
for diff in time_diffs:
|
||||
assert diff >= 5, f"Expected at least 5 seconds between facts, got {diff}"
|
||||
assert diff > 0, f"Expected positive time difference between facts, got {diff}"
|
||||
|
||||
# Update agent_facts to be sorted for subsequent checks
|
||||
agent_facts = sorted_facts
|
||||
|
||||
@@ -7,6 +7,7 @@ Requires Docker to be running. Tests are skipped automatically if Docker is unav
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
import tempfile
|
||||
import time
|
||||
@@ -25,8 +26,12 @@ try:
|
||||
except ImportError:
|
||||
_has_testcontainers = False
|
||||
|
||||
_in_ci = os.getenv("CI") == "true"
|
||||
|
||||
pytestmark = [
|
||||
pytest.mark.skipif(not _has_testcontainers, reason="testcontainers not installed"),
|
||||
pytest.mark.skipif(_in_ci, reason="SeaweedFS Docker image pull too slow in CI"),
|
||||
pytest.mark.timeout(300),
|
||||
]
|
||||
|
||||
SEAWEEDFS_S3_PORT = 8333
|
||||
@@ -105,7 +110,7 @@ def seaweedfs_container():
|
||||
port = container.get_exposed_port(SEAWEEDFS_S3_PORT)
|
||||
endpoint = f"http://{host}:{port}"
|
||||
|
||||
_wait_for_seaweedfs(endpoint)
|
||||
_wait_for_seaweedfs(endpoint, timeout=240)
|
||||
|
||||
# Create test bucket using obstore (proper SigV4 signing)
|
||||
import obstore as obs
|
||||
|
||||
@@ -0,0 +1,171 @@
|
||||
"""
|
||||
Tests for server-side filtering in the graph API endpoint.
|
||||
|
||||
Verifies that q (text search) and tags filters work correctly
|
||||
when passed as query parameters to GET /v1/default/banks/{bank_id}/graph.
|
||||
"""
|
||||
from datetime import datetime
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
import pytest_asyncio
|
||||
|
||||
from hindsight_api.api import create_app
|
||||
|
||||
|
||||
@pytest_asyncio.fixture
|
||||
async def api_client(memory):
|
||||
"""Create an async test client for the FastAPI app."""
|
||||
app = create_app(memory, initialize_memory=False)
|
||||
transport = httpx.ASGITransport(app=app)
|
||||
async with httpx.AsyncClient(transport=transport, base_url="http://test") as client:
|
||||
yield client
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def test_bank_id():
|
||||
"""Provide a unique bank ID for this test run."""
|
||||
return f"graph_filter_test_{datetime.now().timestamp()}"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_graph_no_filter_returns_all(api_client, test_bank_id):
|
||||
"""Without filters the graph endpoint returns all memories."""
|
||||
response = await api_client.post(
|
||||
f"/v1/default/banks/{test_bank_id}/memories",
|
||||
json={
|
||||
"items": [
|
||||
{"content": "Alice loves hiking in the mountains.", "tags": ["user_alice"]},
|
||||
{"content": "Bob enjoys swimming at the beach.", "tags": ["user_bob"]},
|
||||
]
|
||||
},
|
||||
)
|
||||
assert response.status_code == 200
|
||||
|
||||
response = await api_client.get(f"/v1/default/banks/{test_bank_id}/graph")
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
assert "table_rows" in data
|
||||
texts = [row["text"] for row in data["table_rows"]]
|
||||
assert any("Alice" in t for t in texts)
|
||||
assert any("Bob" in t for t in texts)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_graph_q_filter_returns_matching(api_client, test_bank_id):
|
||||
"""The q parameter filters memories by text content."""
|
||||
response = await api_client.post(
|
||||
f"/v1/default/banks/{test_bank_id}/memories",
|
||||
json={
|
||||
"items": [
|
||||
{"content": "Alice loves hiking in the mountains."},
|
||||
{"content": "Bob enjoys swimming at the beach."},
|
||||
]
|
||||
},
|
||||
)
|
||||
assert response.status_code == 200
|
||||
|
||||
response = await api_client.get(f"/v1/default/banks/{test_bank_id}/graph", params={"q": "Alice"})
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
texts = [row["text"] for row in data["table_rows"]]
|
||||
assert all("Alice" in t or "alice" in t.lower() for t in texts), (
|
||||
f"Expected only Alice memories, got: {texts}"
|
||||
)
|
||||
assert not any("Bob" in t for t in texts)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_graph_q_filter_case_insensitive(api_client, test_bank_id):
|
||||
"""The q filter is case-insensitive."""
|
||||
response = await api_client.post(
|
||||
f"/v1/default/banks/{test_bank_id}/memories",
|
||||
json={
|
||||
"items": [
|
||||
{"content": "Alice loves hiking in the mountains."},
|
||||
{"content": "Bob enjoys swimming at the beach."},
|
||||
]
|
||||
},
|
||||
)
|
||||
assert response.status_code == 200
|
||||
|
||||
response = await api_client.get(f"/v1/default/banks/{test_bank_id}/graph", params={"q": "alice"})
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
texts = [row["text"] for row in data["table_rows"]]
|
||||
assert any("Alice" in t for t in texts)
|
||||
assert not any("Bob" in t for t in texts)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_graph_tags_filter_returns_matching(api_client, test_bank_id):
|
||||
"""The tags parameter filters memories to only those with matching tags."""
|
||||
response = await api_client.post(
|
||||
f"/v1/default/banks/{test_bank_id}/memories",
|
||||
json={
|
||||
"items": [
|
||||
{"content": "Alice loves hiking.", "tags": ["user_alice"]},
|
||||
{"content": "Bob enjoys swimming.", "tags": ["user_bob"]},
|
||||
]
|
||||
},
|
||||
)
|
||||
assert response.status_code == 200
|
||||
|
||||
response = await api_client.get(
|
||||
f"/v1/default/banks/{test_bank_id}/graph",
|
||||
params={"tags": "user_alice", "tags_match": "all_strict"},
|
||||
)
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
texts = [row["text"] for row in data["table_rows"]]
|
||||
assert any("Alice" in t for t in texts)
|
||||
assert not any("Bob" in t for t in texts)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_graph_q_and_tags_filter_combined(api_client, test_bank_id):
|
||||
"""Combining q and tags filters applies both server-side."""
|
||||
response = await api_client.post(
|
||||
f"/v1/default/banks/{test_bank_id}/memories",
|
||||
json={
|
||||
"items": [
|
||||
{"content": "Alice loves hiking.", "tags": ["user_alice"]},
|
||||
{"content": "Alice also loves coding.", "tags": ["user_alice"]},
|
||||
{"content": "Bob enjoys swimming.", "tags": ["user_bob"]},
|
||||
]
|
||||
},
|
||||
)
|
||||
assert response.status_code == 200
|
||||
|
||||
response = await api_client.get(
|
||||
f"/v1/default/banks/{test_bank_id}/graph",
|
||||
params={"q": "hiking", "tags": "user_alice", "tags_match": "all_strict"},
|
||||
)
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
texts = [row["text"] for row in data["table_rows"]]
|
||||
assert any("hiking" in t.lower() for t in texts)
|
||||
assert not any("coding" in t.lower() for t in texts)
|
||||
assert not any("Bob" in t for t in texts)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_graph_q_filter_empty_results(api_client, test_bank_id):
|
||||
"""The q filter returns empty results when no memory matches."""
|
||||
response = await api_client.post(
|
||||
f"/v1/default/banks/{test_bank_id}/memories",
|
||||
json={
|
||||
"items": [
|
||||
{"content": "Alice loves hiking."},
|
||||
]
|
||||
},
|
||||
)
|
||||
assert response.status_code == 200
|
||||
|
||||
response = await api_client.get(
|
||||
f"/v1/default/banks/{test_bank_id}/graph",
|
||||
params={"q": "zzznomatchzzz"},
|
||||
)
|
||||
assert response.status_code == 200
|
||||
data = response.json()
|
||||
assert data["table_rows"] == []
|
||||
@@ -15,9 +15,6 @@ from hindsight_api.config_resolver import ConfigResolver
|
||||
from hindsight_api.extensions.tenant import TenantExtension
|
||||
from hindsight_api.models import RequestContext
|
||||
|
||||
# Enable bank config API for all tests in this module
|
||||
os.environ["HINDSIGHT_API_ENABLE_BANK_CONFIG_API"] = "true"
|
||||
|
||||
|
||||
class MockTenantExtension(TenantExtension):
|
||||
"""Mock tenant extension for testing tenant-level config."""
|
||||
@@ -74,12 +71,22 @@ async def test_hierarchical_fields_categorization():
|
||||
|
||||
# Verify configurable fields include behavioral settings (safe to modify)
|
||||
assert "retain_extraction_mode" in configurable
|
||||
assert "enable_observations" in configurable
|
||||
assert "retain_chunk_size" in configurable
|
||||
assert "retain_mission" in configurable
|
||||
assert "retain_custom_instructions" in configurable
|
||||
assert "retain_chunk_size" in configurable
|
||||
assert "enable_observations" in configurable
|
||||
assert "observations_mission" in configurable
|
||||
assert "reflect_mission" in configurable
|
||||
assert "disposition_skepticism" in configurable
|
||||
assert "disposition_literalism" in configurable
|
||||
assert "disposition_empathy" in configurable
|
||||
|
||||
# Verify count is correct (only 4 fields)
|
||||
assert len(configurable) == 4
|
||||
# Verify entity labels fields are included
|
||||
assert "entities_allow_free_form" in configurable
|
||||
assert "entity_labels" in configurable
|
||||
|
||||
# Verify count is correct
|
||||
assert len(configurable) == 13
|
||||
|
||||
# Verify credential fields (NEVER exposed)
|
||||
assert "llm_api_key" in credentials
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
"""
|
||||
Integration tests for the Iris file parser.
|
||||
|
||||
Tests are skipped automatically if HINDSIGHT_API_FILE_PARSER_IRIS_TOKEN
|
||||
and HINDSIGHT_API_FILE_PARSER_IRIS_ORG_ID are not set in the environment.
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from hindsight_api.config import ENV_FILE_PARSER_IRIS_ORG_ID, ENV_FILE_PARSER_IRIS_TOKEN
|
||||
from hindsight_api.engine.parsers.iris import IrisParser
|
||||
|
||||
_token = os.getenv(ENV_FILE_PARSER_IRIS_TOKEN)
|
||||
_org_id = os.getenv(ENV_FILE_PARSER_IRIS_ORG_ID)
|
||||
|
||||
pytestmark = pytest.mark.skipif(
|
||||
not (_token and _org_id),
|
||||
reason="HINDSIGHT_API_FILE_PARSER_IRIS_TOKEN and HINDSIGHT_API_FILE_PARSER_IRIS_ORG_ID not set",
|
||||
)
|
||||
|
||||
# Minimal valid PDF with the text "Hello from Hindsight"
|
||||
_SAMPLE_PDF = b"""%PDF-1.4
|
||||
1 0 obj
|
||||
<< /Type /Catalog /Pages 2 0 R >>
|
||||
endobj
|
||||
2 0 obj
|
||||
<< /Type /Pages /Kids [3 0 R] /Count 1 >>
|
||||
endobj
|
||||
3 0 obj
|
||||
<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792]
|
||||
/Contents 4 0 R /Resources << /Font << /F1 << /Type /Font /Subtype /Type1 /BaseFont /Helvetica >> >> >> >>
|
||||
endobj
|
||||
4 0 obj
|
||||
<< /Length 44 >>
|
||||
stream
|
||||
BT /F1 12 Tf 100 700 Td (Hello from Hindsight) Tj ET
|
||||
endstream
|
||||
endobj
|
||||
xref
|
||||
0 5
|
||||
0000000000 65535 f
|
||||
0000000009 00000 n
|
||||
0000000058 00000 n
|
||||
0000000115 00000 n
|
||||
0000000274 00000 n
|
||||
trailer << /Size 5 /Root 1 0 R >>
|
||||
startxref
|
||||
369
|
||||
%%EOF"""
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def iris_parser() -> IrisParser:
|
||||
return IrisParser(token=_token, org_id=_org_id)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_iris_parser_converts_pdf(iris_parser: IrisParser):
|
||||
"""IrisParser should extract text from a valid PDF."""
|
||||
result = await iris_parser.convert(_SAMPLE_PDF, "sample.pdf")
|
||||
assert isinstance(result, str)
|
||||
assert len(result) > 0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_iris_parser_name(iris_parser: IrisParser):
|
||||
"""IrisParser.name() should return 'iris'."""
|
||||
assert iris_parser.name() == "iris"
|
||||
|
||||
|
||||
@@ -85,6 +85,7 @@ class TestLiteLLMSDKEmbeddings:
|
||||
model="cohere/embed-english-v3.0",
|
||||
input=["test"],
|
||||
api_key="test_key",
|
||||
encoding_format="float",
|
||||
)
|
||||
|
||||
async def test_initialization_missing_package(self):
|
||||
@@ -137,6 +138,7 @@ class TestLiteLLMSDKEmbeddings:
|
||||
model="cohere/embed-english-v3.0",
|
||||
input=["Hello world"],
|
||||
api_key="test_key",
|
||||
encoding_format="float",
|
||||
)
|
||||
|
||||
async def test_encode_multiple_texts(self, embeddings, mock_litellm):
|
||||
|
||||
@@ -226,6 +226,7 @@ async def test_llm_provider_api_methods(provider: str, model: str):
|
||||
|
||||
@pytest.mark.parametrize("provider,model", MODEL_MATRIX)
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.timeout(300)
|
||||
async def test_llm_provider_memory_operations(provider: str, model: str):
|
||||
"""
|
||||
Test LLM provider with actual memory operations: fact extraction and reflect.
|
||||
@@ -331,8 +332,8 @@ async def test_llm_provider_consolidation(memory_no_llm_verify, request_context,
|
||||
test_bank_id = f"llm_test_consolidation_{provider}_{model}_{datetime.now().timestamp()}"
|
||||
|
||||
# Enable observations for this bank
|
||||
from hindsight_api.config import get_config
|
||||
config = get_config()
|
||||
from hindsight_api.config import _get_raw_config
|
||||
config = _get_raw_config()
|
||||
original_value = config.enable_observations
|
||||
config.enable_observations = True
|
||||
|
||||
|
||||
@@ -165,5 +165,5 @@ class TestMCPExtensionIntegration:
|
||||
assert "create_bank" in tools
|
||||
# Extension tool also present
|
||||
assert "test_extension_tool" in tools
|
||||
# At least 11 core + 1 extension = 12 tools (may grow as new tools are added)
|
||||
assert len(tools) >= 12
|
||||
# At least 29 core + 1 extension = 30 tools (may grow as new tools are added)
|
||||
assert len(tools) >= 30
|
||||
|
||||
@@ -1,212 +0,0 @@
|
||||
"""Test local MCP server."""
|
||||
|
||||
import asyncio
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_memory():
|
||||
"""Create a mock MemoryEngine."""
|
||||
memory = MagicMock()
|
||||
memory._initialized = True
|
||||
memory.retain_batch_async = AsyncMock()
|
||||
memory.recall_async = AsyncMock(return_value=MagicMock(results=[]))
|
||||
return memory
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_local_mcp_server_retain(mock_memory):
|
||||
"""Test that retain tool fires async and returns immediately."""
|
||||
from hindsight_api.mcp_local import create_local_mcp_server
|
||||
|
||||
bank_id = "test-bank"
|
||||
mcp_server = create_local_mcp_server(bank_id, memory=mock_memory)
|
||||
|
||||
# Get the tools
|
||||
tools = mcp_server._tool_manager._tools
|
||||
assert "retain" in tools
|
||||
|
||||
# Call retain
|
||||
retain_tool = tools["retain"]
|
||||
result = await retain_tool.fn(content="test content", context="test_context")
|
||||
|
||||
# Returns immediately with accepted status
|
||||
assert result["status"] == "accepted"
|
||||
|
||||
# Wait for background task to complete
|
||||
await asyncio.sleep(0.1)
|
||||
|
||||
# Verify the memory was called correctly
|
||||
mock_memory.retain_batch_async.assert_called_once()
|
||||
call_kwargs = mock_memory.retain_batch_async.call_args.kwargs
|
||||
assert call_kwargs["bank_id"] == "test-bank"
|
||||
assert call_kwargs["contents"] == [{"content": "test content", "context": "test_context"}]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_local_mcp_server_recall(mock_memory):
|
||||
"""Test that recall tool calls memory.recall_async with correct params."""
|
||||
from hindsight_api.mcp_local import create_local_mcp_server
|
||||
from hindsight_api.engine.memory_engine import Budget
|
||||
|
||||
# Mock recall_async to return a proper pydantic model
|
||||
mock_result = MagicMock()
|
||||
mock_result.model_dump.return_value = {"results": []}
|
||||
mock_memory.recall_async = AsyncMock(return_value=mock_result)
|
||||
|
||||
bank_id = "test-bank"
|
||||
mcp_server = create_local_mcp_server(bank_id, memory=mock_memory)
|
||||
|
||||
# Get the tools
|
||||
tools = mcp_server._tool_manager._tools
|
||||
assert "recall" in tools
|
||||
|
||||
# Call recall
|
||||
recall_tool = tools["recall"]
|
||||
result = await recall_tool.fn(query="test query", max_tokens=2048)
|
||||
|
||||
# Result is a dict
|
||||
assert isinstance(result, dict)
|
||||
|
||||
# Verify the memory was called correctly
|
||||
mock_memory.recall_async.assert_called_once()
|
||||
call_kwargs = mock_memory.recall_async.call_args.kwargs
|
||||
assert call_kwargs["bank_id"] == "test-bank"
|
||||
assert call_kwargs["query"] == "test query"
|
||||
assert call_kwargs["max_tokens"] == 2048
|
||||
assert call_kwargs["budget"] == Budget.HIGH
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_local_mcp_server_retain_with_default_context(mock_memory):
|
||||
"""Test that retain uses default context when not provided."""
|
||||
from hindsight_api.mcp_local import create_local_mcp_server
|
||||
|
||||
bank_id = "test-bank"
|
||||
mcp_server = create_local_mcp_server(bank_id, memory=mock_memory)
|
||||
|
||||
tools = mcp_server._tool_manager._tools
|
||||
retain_tool = tools["retain"]
|
||||
|
||||
# Call retain without context
|
||||
await retain_tool.fn(content="test content")
|
||||
|
||||
# Wait for background task
|
||||
await asyncio.sleep(0.1)
|
||||
|
||||
call_kwargs = mock_memory.retain_batch_async.call_args.kwargs
|
||||
assert call_kwargs["contents"] == [{"content": "test content", "context": "general"}]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_local_mcp_server_retain_error_handling(mock_memory):
|
||||
"""Test that retain errors are logged but don't affect response."""
|
||||
from hindsight_api.mcp_local import create_local_mcp_server
|
||||
|
||||
mock_memory.retain_batch_async = AsyncMock(side_effect=Exception("Test error"))
|
||||
|
||||
mcp_server = create_local_mcp_server("test-bank", memory=mock_memory)
|
||||
|
||||
tools = mcp_server._tool_manager._tools
|
||||
retain_tool = tools["retain"]
|
||||
|
||||
# Retain returns immediately with accepted status (fire and forget)
|
||||
result = await retain_tool.fn(content="test content")
|
||||
assert result["status"] == "accepted"
|
||||
|
||||
# Wait for background task to complete (and log error)
|
||||
await asyncio.sleep(0.1)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_local_mcp_server_recall_error_handling(mock_memory):
|
||||
"""Test that recall handles errors gracefully."""
|
||||
from hindsight_api.mcp_local import create_local_mcp_server
|
||||
|
||||
mock_memory.recall_async = AsyncMock(side_effect=Exception("Test error"))
|
||||
|
||||
mcp_server = create_local_mcp_server("test-bank", memory=mock_memory)
|
||||
|
||||
tools = mcp_server._tool_manager._tools
|
||||
recall_tool = tools["recall"]
|
||||
|
||||
result = await recall_tool.fn(query="test query")
|
||||
|
||||
# Result is a dict with error
|
||||
assert isinstance(result, dict)
|
||||
assert "error" in result
|
||||
assert result["results"] == []
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_local_mcp_server_recall_with_defaults(mock_memory):
|
||||
"""Test that recall uses default max_tokens and HIGH budget."""
|
||||
from hindsight_api.mcp_local import create_local_mcp_server
|
||||
from hindsight_api.engine.memory_engine import Budget
|
||||
|
||||
mock_result = MagicMock()
|
||||
mock_result.model_dump.return_value = {"results": []}
|
||||
mock_memory.recall_async = AsyncMock(return_value=mock_result)
|
||||
|
||||
mcp_server = create_local_mcp_server("test-bank", memory=mock_memory)
|
||||
|
||||
tools = mcp_server._tool_manager._tools
|
||||
recall_tool = tools["recall"]
|
||||
|
||||
# Call with defaults
|
||||
await recall_tool.fn(query="test query")
|
||||
|
||||
call_kwargs = mock_memory.recall_async.call_args.kwargs
|
||||
assert call_kwargs["max_tokens"] == 4096
|
||||
assert call_kwargs["budget"] == Budget.HIGH
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_local_mcp_server_retain_with_timestamp(mock_memory):
|
||||
"""Test that retain passes timestamp as event_date."""
|
||||
from datetime import datetime, timezone
|
||||
from hindsight_api.mcp_local import create_local_mcp_server
|
||||
|
||||
mcp_server = create_local_mcp_server("test-bank", memory=mock_memory)
|
||||
|
||||
tools = mcp_server._tool_manager._tools
|
||||
retain_tool = tools["retain"]
|
||||
|
||||
# Call retain with timestamp
|
||||
result = await retain_tool.fn(
|
||||
content="test content", context="test_context", timestamp="2024-01-15T10:30:00Z"
|
||||
)
|
||||
|
||||
assert result["status"] == "accepted"
|
||||
|
||||
# Wait for background task
|
||||
await asyncio.sleep(0.1)
|
||||
|
||||
call_kwargs = mock_memory.retain_batch_async.call_args.kwargs
|
||||
contents = call_kwargs["contents"]
|
||||
assert len(contents) == 1
|
||||
assert contents[0]["content"] == "test content"
|
||||
assert contents[0]["context"] == "test_context"
|
||||
assert "event_date" in contents[0]
|
||||
assert contents[0]["event_date"] == datetime(2024, 1, 15, 10, 30, 0, tzinfo=timezone.utc)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_local_mcp_server_retain_with_invalid_timestamp(mock_memory):
|
||||
"""Test that retain rejects invalid timestamp format."""
|
||||
from hindsight_api.mcp_local import create_local_mcp_server
|
||||
|
||||
mcp_server = create_local_mcp_server("test-bank", memory=mock_memory)
|
||||
|
||||
tools = mcp_server._tool_manager._tools
|
||||
retain_tool = tools["retain"]
|
||||
|
||||
# Call retain with invalid timestamp
|
||||
result = await retain_tool.fn(content="test content", timestamp="not-a-date")
|
||||
|
||||
assert result["status"] == "error"
|
||||
assert "Invalid timestamp format" in result["message"]
|
||||
|
||||
# Verify retain_batch_async was NOT called
|
||||
mock_memory.retain_batch_async.assert_not_called()
|
||||
@@ -46,16 +46,15 @@ async def test_mcp_tools_use_context_bank_id(mock_memory):
|
||||
assert "retain" in tools
|
||||
assert "recall" in tools
|
||||
|
||||
# Test retain with bank_id from context (use async_processing=False for synchronous test)
|
||||
token = _current_bank_id.set("context-bank-id")
|
||||
try:
|
||||
retain_tool = tools["retain"]
|
||||
result = await retain_tool.fn(content="test content", context="test_context", async_processing=False)
|
||||
assert "successfully" in result.lower()
|
||||
result = await retain_tool.fn(content="test content", context="test_context")
|
||||
assert result["status"] == "accepted"
|
||||
|
||||
# Verify the memory was called with the context bank_id
|
||||
mock_memory.retain_batch_async.assert_called_once()
|
||||
call_kwargs = mock_memory.retain_batch_async.call_args.kwargs
|
||||
mock_memory.submit_async_retain.assert_called_once()
|
||||
call_kwargs = mock_memory.submit_async_retain.call_args.kwargs
|
||||
assert call_kwargs["bank_id"] == "context-bank-id"
|
||||
finally:
|
||||
_current_bank_id.reset(token)
|
||||
@@ -133,12 +132,12 @@ async def test_mcp_tools_propagate_api_key(mock_memory):
|
||||
api_key_token = _current_api_key.set("test-bearer-token")
|
||||
try:
|
||||
retain_tool = tools["retain"]
|
||||
result = await retain_tool.fn(content="test content", context="test_context", async_processing=False)
|
||||
assert "successfully" in result.lower()
|
||||
result = await retain_tool.fn(content="test content", context="test_context")
|
||||
assert result["status"] == "accepted"
|
||||
|
||||
# Verify the memory was called with request_context containing api_key
|
||||
mock_memory.retain_batch_async.assert_called_once()
|
||||
call_kwargs = mock_memory.retain_batch_async.call_args.kwargs
|
||||
mock_memory.submit_async_retain.assert_called_once()
|
||||
call_kwargs = mock_memory.submit_async_retain.call_args.kwargs
|
||||
assert call_kwargs["request_context"].api_key == "test-bearer-token"
|
||||
finally:
|
||||
_current_bank_id.reset(bank_token)
|
||||
@@ -200,11 +199,11 @@ async def test_mcp_tools_propagate_tenant_id_and_api_key_id(mock_memory):
|
||||
key_id_token = _current_api_key_id.set("key-uuid-456")
|
||||
try:
|
||||
retain_tool = tools["retain"]
|
||||
await retain_tool.fn(content="test content", context="test_context", async_processing=False)
|
||||
await retain_tool.fn(content="test content", context="test_context")
|
||||
|
||||
# Verify the RequestContext passed to memory engine has all auth fields
|
||||
mock_memory.retain_batch_async.assert_called_once()
|
||||
request_context = mock_memory.retain_batch_async.call_args.kwargs["request_context"]
|
||||
mock_memory.submit_async_retain.assert_called_once()
|
||||
request_context = mock_memory.submit_async_retain.call_args.kwargs["request_context"]
|
||||
assert request_context.api_key == "hsk_test_key"
|
||||
assert request_context.tenant_id == "org-billing-123"
|
||||
assert request_context.api_key_id == "key-uuid-456"
|
||||
@@ -352,6 +351,69 @@ async def test_middleware_handles_both_endpoints(mock_memory):
|
||||
assert "create_bank" not in single_bank_tools
|
||||
|
||||
|
||||
def test_global_mcp_enabled_tools_filter_restricts_registered_tools(mock_memory):
|
||||
"""Test that global mcp_enabled_tools env setting restricts which tools are registered."""
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
from hindsight_api.api.mcp import create_mcp_server
|
||||
|
||||
mock_cfg = MagicMock()
|
||||
mock_cfg.mcp_enabled_tools = ["retain", "recall"]
|
||||
|
||||
with patch("hindsight_api.api.mcp._get_raw_config", return_value=mock_cfg):
|
||||
mcp_server = create_mcp_server(mock_memory, multi_bank=True)
|
||||
|
||||
tools = mcp_server._tool_manager._tools
|
||||
assert "retain" in tools
|
||||
assert "recall" in tools
|
||||
assert "reflect" not in tools
|
||||
assert "list_banks" not in tools
|
||||
assert "create_bank" not in tools
|
||||
assert "list_mental_models" not in tools
|
||||
|
||||
|
||||
def test_global_mcp_enabled_tools_none_exposes_all_tools(mock_memory):
|
||||
"""Test that mcp_enabled_tools=None (default) exposes all tools."""
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
from hindsight_api.api.mcp import create_mcp_server
|
||||
|
||||
mock_cfg = MagicMock()
|
||||
mock_cfg.mcp_enabled_tools = None
|
||||
|
||||
with patch("hindsight_api.api.mcp._get_raw_config", return_value=mock_cfg):
|
||||
mcp_server = create_mcp_server(mock_memory, multi_bank=True)
|
||||
|
||||
tools = mcp_server._tool_manager._tools
|
||||
assert "retain" in tools
|
||||
assert "recall" in tools
|
||||
assert "reflect" in tools
|
||||
assert "list_banks" in tools
|
||||
assert "create_bank" in tools
|
||||
|
||||
|
||||
def test_global_mcp_enabled_tools_intersects_with_single_bank_mode(mock_memory):
|
||||
"""Test that global filter intersects with single-bank mode tool set.
|
||||
|
||||
list_banks is in the global allowlist but NOT in single-bank mode, so it
|
||||
should be absent from the final registered set.
|
||||
"""
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
from hindsight_api.api.mcp import create_mcp_server
|
||||
|
||||
mock_cfg = MagicMock()
|
||||
mock_cfg.mcp_enabled_tools = ["retain", "recall", "list_banks"]
|
||||
|
||||
with patch("hindsight_api.api.mcp._get_raw_config", return_value=mock_cfg):
|
||||
mcp_server = create_mcp_server(mock_memory, multi_bank=False)
|
||||
|
||||
tools = mcp_server._tool_manager._tools
|
||||
assert "retain" in tools
|
||||
assert "recall" in tools
|
||||
assert "list_banks" not in tools # single-bank mode excludes it regardless
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_routing_logic_from_url_path():
|
||||
"""Test that routing correctly selects server based on URL structure.
|
||||
|
||||
@@ -77,8 +77,9 @@ class TestBuildContentDict:
|
||||
|
||||
@pytest.fixture
|
||||
def mock_memory():
|
||||
"""Create a mock MemoryEngine with mental model methods."""
|
||||
"""Create a mock MemoryEngine with all MCP tool methods."""
|
||||
memory = MagicMock()
|
||||
# Mental model methods
|
||||
memory.list_mental_models = AsyncMock(
|
||||
return_value=[
|
||||
{"id": "mm-1", "name": "Coding Prefs", "source_query": "coding preferences?", "content": "Prefers Python"},
|
||||
@@ -104,6 +105,41 @@ def mock_memory():
|
||||
}
|
||||
)
|
||||
memory.delete_mental_model = AsyncMock(return_value=True)
|
||||
|
||||
# Retain/recall/reflect
|
||||
memory.retain_batch_async = AsyncMock()
|
||||
memory.submit_async_retain = AsyncMock(return_value={"operation_id": "op-retain"})
|
||||
memory.recall_async = AsyncMock(return_value=MagicMock(model_dump_json=lambda indent=None: '{"results": []}', model_dump=lambda: {"results": []}))
|
||||
memory.reflect_async = AsyncMock(return_value=MagicMock(model_dump_json=lambda indent=None: '{"text": "reflection"}', model_dump=lambda: {"text": "reflection"}, structured_output=None))
|
||||
|
||||
# Directive methods
|
||||
memory.list_directives = AsyncMock(return_value=[{"id": "dir-1", "name": "Be concise", "content": "Keep responses short"}])
|
||||
memory.create_directive = AsyncMock(return_value={"id": "dir-new", "name": "Test", "content": "Test content"})
|
||||
memory.delete_directive = AsyncMock(return_value=True)
|
||||
|
||||
# Memory browsing methods
|
||||
memory.list_memory_units = AsyncMock(return_value={"items": [{"id": "mem-1", "content": "Test"}], "total": 1})
|
||||
memory.get_memory_unit = AsyncMock(return_value={"id": "mem-1", "content": "Test memory"})
|
||||
memory.delete_memory_unit = AsyncMock(return_value={"deleted_count": 1})
|
||||
|
||||
# Document methods
|
||||
memory.list_documents = AsyncMock(return_value={"items": [{"id": "doc-1", "name": "Test Doc"}], "total": 1})
|
||||
memory.get_document = AsyncMock(return_value={"id": "doc-1", "name": "Test Doc"})
|
||||
memory.delete_document = AsyncMock(return_value={"deleted_memories": 5})
|
||||
|
||||
# Operation methods
|
||||
memory.list_operations = AsyncMock(return_value={"items": [{"id": "op-1", "status": "completed"}]})
|
||||
memory.get_operation_status = AsyncMock(return_value={"id": "op-1", "status": "completed", "progress": 100})
|
||||
memory.cancel_operation = AsyncMock(return_value={"id": "op-1", "status": "cancelled"})
|
||||
|
||||
# Tags & bank methods
|
||||
memory.list_tags = AsyncMock(return_value={"items": ["tag1", "tag2"], "total": 2})
|
||||
memory.get_bank_profile = AsyncMock(return_value={"id": "test-bank", "name": "Test Bank", "mission": "Testing"})
|
||||
memory.get_bank_stats = AsyncMock(return_value={"nodes": 100, "links": 50})
|
||||
memory.update_bank = AsyncMock(return_value={"id": "test-bank", "name": "Updated"})
|
||||
memory.delete_bank = AsyncMock(return_value={"deleted_memories": 10, "deleted_entities": 5})
|
||||
memory.list_banks = AsyncMock(return_value=[])
|
||||
|
||||
return memory
|
||||
|
||||
|
||||
@@ -211,7 +247,7 @@ class TestMentalModelToolRegistration:
|
||||
assert request_context.api_key == "test-api-key"
|
||||
|
||||
def test_mental_model_tools_in_default_set(self):
|
||||
"""Mental model tools should be in the default tools set when config.tools is None."""
|
||||
"""All tools should be in the default tools set when config.tools is None."""
|
||||
from fastmcp import FastMCP
|
||||
|
||||
memory = MagicMock()
|
||||
@@ -229,6 +265,21 @@ class TestMentalModelToolRegistration:
|
||||
memory.submit_async_refresh_mental_model = AsyncMock()
|
||||
memory.update_mental_model = AsyncMock()
|
||||
memory.delete_mental_model = AsyncMock()
|
||||
memory.list_directives = AsyncMock(return_value=[])
|
||||
memory.create_directive = AsyncMock()
|
||||
memory.delete_directive = AsyncMock()
|
||||
memory.list_memory_units = AsyncMock(return_value={})
|
||||
memory.get_memory_unit = AsyncMock()
|
||||
memory.delete_memory_unit = AsyncMock()
|
||||
memory.list_documents = AsyncMock(return_value={})
|
||||
memory.get_document = AsyncMock()
|
||||
memory.delete_document = AsyncMock()
|
||||
memory.list_operations = AsyncMock(return_value={})
|
||||
memory.get_operation_status = AsyncMock()
|
||||
memory.cancel_operation = AsyncMock()
|
||||
memory.list_tags = AsyncMock(return_value={})
|
||||
memory.get_bank_stats = AsyncMock(return_value={})
|
||||
memory.delete_bank = AsyncMock(return_value={})
|
||||
|
||||
mcp = FastMCP("test", stateless_http=True)
|
||||
config = MCPToolsConfig(
|
||||
@@ -241,6 +292,18 @@ class TestMentalModelToolRegistration:
|
||||
assert "list_mental_models" in tools
|
||||
assert "create_mental_model" in tools
|
||||
assert "refresh_mental_model" in tools
|
||||
# New tools
|
||||
assert "list_directives" in tools
|
||||
assert "list_memories" in tools
|
||||
assert "list_documents" in tools
|
||||
assert "list_operations" in tools
|
||||
assert "list_tags" in tools
|
||||
assert "get_bank" in tools
|
||||
assert "get_bank_stats" in tools
|
||||
assert "update_bank" in tools
|
||||
assert "delete_bank" in tools
|
||||
assert "clear_memories" in tools
|
||||
assert len(tools) == 29
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
@@ -644,3 +707,653 @@ class TestMentalModelInputValidation:
|
||||
result = await _tools(mcp_server_single_bank)["get_mental_model"].fn(mental_model_id="missing")
|
||||
assert isinstance(result, dict)
|
||||
assert "fixed-bank" in result["error"]
|
||||
|
||||
|
||||
# =========================================================================
|
||||
# New Parameter Tests for Existing Tools
|
||||
# =========================================================================
|
||||
|
||||
|
||||
def _make_mcp_server(mock_memory, tools, include_bank_id=True):
|
||||
"""Helper to create an MCP server with specific tools."""
|
||||
from fastmcp import FastMCP
|
||||
|
||||
mcp = FastMCP("test", stateless_http=True)
|
||||
config = MCPToolsConfig(
|
||||
bank_id_resolver=lambda: "test-bank",
|
||||
include_bank_id_param=include_bank_id,
|
||||
tools=tools,
|
||||
)
|
||||
register_mcp_tools(mcp, mock_memory, config)
|
||||
return mcp
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestRetainNewParams:
|
||||
"""Tests for new retain parameters: tags, metadata, document_id."""
|
||||
|
||||
async def test_retain_with_tags(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"retain"})
|
||||
await _tools(mcp)["retain"].fn(content="test", tags=["user:123", "project:alpha"])
|
||||
call_args = mock_memory.submit_async_retain.call_args
|
||||
contents = call_args.kwargs["contents"]
|
||||
assert contents[0]["tags"] == ["user:123", "project:alpha"]
|
||||
|
||||
async def test_retain_with_metadata(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"retain"})
|
||||
await _tools(mcp)["retain"].fn(content="test", metadata={"source": "slack"})
|
||||
call_args = mock_memory.submit_async_retain.call_args
|
||||
contents = call_args.kwargs["contents"]
|
||||
assert contents[0]["metadata"] == {"source": "slack"}
|
||||
|
||||
async def test_retain_with_document_id(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"retain"})
|
||||
await _tools(mcp)["retain"].fn(content="test", document_id="doc-1")
|
||||
call_args = mock_memory.submit_async_retain.call_args
|
||||
contents = call_args.kwargs["contents"]
|
||||
assert contents[0]["document_id"] == "doc-1"
|
||||
|
||||
async def test_retain_without_new_params_backward_compat(self, mock_memory):
|
||||
"""Existing behavior preserved when new params not provided."""
|
||||
mcp = _make_mcp_server(mock_memory, {"retain"})
|
||||
await _tools(mcp)["retain"].fn(content="test")
|
||||
call_args = mock_memory.submit_async_retain.call_args
|
||||
contents = call_args.kwargs["contents"]
|
||||
assert "tags" not in contents[0]
|
||||
assert "metadata" not in contents[0]
|
||||
assert "document_id" not in contents[0]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestRecallNewParams:
|
||||
"""Tests for new recall parameters: budget, types, tags, tags_match, query_timestamp."""
|
||||
|
||||
async def test_recall_default_budget_high(self, mock_memory):
|
||||
"""Default budget should be HIGH (backward compat)."""
|
||||
from hindsight_api.engine.memory_engine import Budget
|
||||
|
||||
mcp = _make_mcp_server(mock_memory, {"recall"})
|
||||
await _tools(mcp)["recall"].fn(query="test")
|
||||
call_kwargs = mock_memory.recall_async.call_args.kwargs
|
||||
assert call_kwargs["budget"] == Budget.HIGH
|
||||
|
||||
async def test_recall_budget_low(self, mock_memory):
|
||||
from hindsight_api.engine.memory_engine import Budget
|
||||
|
||||
mcp = _make_mcp_server(mock_memory, {"recall"})
|
||||
await _tools(mcp)["recall"].fn(query="test", budget="low")
|
||||
call_kwargs = mock_memory.recall_async.call_args.kwargs
|
||||
assert call_kwargs["budget"] == Budget.LOW
|
||||
|
||||
async def test_recall_with_types(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"recall"})
|
||||
await _tools(mcp)["recall"].fn(query="test", types=["world"])
|
||||
call_kwargs = mock_memory.recall_async.call_args.kwargs
|
||||
assert call_kwargs["fact_type"] == ["world"]
|
||||
|
||||
async def test_recall_default_types_all(self, mock_memory):
|
||||
from hindsight_api.engine.response_models import VALID_RECALL_FACT_TYPES
|
||||
|
||||
mcp = _make_mcp_server(mock_memory, {"recall"})
|
||||
await _tools(mcp)["recall"].fn(query="test")
|
||||
call_kwargs = mock_memory.recall_async.call_args.kwargs
|
||||
assert call_kwargs["fact_type"] == list(VALID_RECALL_FACT_TYPES)
|
||||
|
||||
async def test_recall_with_tags(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"recall"})
|
||||
await _tools(mcp)["recall"].fn(query="test", tags=["project:x"])
|
||||
call_kwargs = mock_memory.recall_async.call_args.kwargs
|
||||
assert call_kwargs["tags"] == ["project:x"]
|
||||
assert call_kwargs["tags_match"] == "any"
|
||||
|
||||
async def test_recall_with_query_timestamp(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"recall"})
|
||||
await _tools(mcp)["recall"].fn(query="test", query_timestamp="2024-01-01T00:00:00Z")
|
||||
call_kwargs = mock_memory.recall_async.call_args.kwargs
|
||||
assert call_kwargs["question_date"] == datetime(2024, 1, 1, 0, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestReflectNewParams:
|
||||
"""Tests for new reflect parameters: max_tokens, response_schema, tags, tags_match."""
|
||||
|
||||
async def test_reflect_with_max_tokens(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"reflect"})
|
||||
await _tools(mcp)["reflect"].fn(query="test", max_tokens=2048)
|
||||
call_kwargs = mock_memory.reflect_async.call_args.kwargs
|
||||
assert call_kwargs["max_tokens"] == 2048
|
||||
|
||||
async def test_reflect_with_response_schema(self, mock_memory):
|
||||
schema = {"type": "object", "properties": {"answer": {"type": "string"}}}
|
||||
mock_memory.reflect_async = AsyncMock(
|
||||
return_value=MagicMock(
|
||||
model_dump_json=lambda indent=None: '{"text": "reflection"}',
|
||||
model_dump=lambda: {"text": "reflection"},
|
||||
structured_output={"answer": "yes"},
|
||||
)
|
||||
)
|
||||
mcp = _make_mcp_server(mock_memory, {"reflect"})
|
||||
result = await _tools(mcp)["reflect"].fn(query="test", response_schema=schema)
|
||||
call_kwargs = mock_memory.reflect_async.call_args.kwargs
|
||||
assert call_kwargs["response_schema"] == schema
|
||||
# Multi-bank returns JSON string
|
||||
import json
|
||||
|
||||
parsed = json.loads(result)
|
||||
assert parsed["structured_output"] == {"answer": "yes"}
|
||||
|
||||
async def test_reflect_with_tags(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"reflect"})
|
||||
await _tools(mcp)["reflect"].fn(query="test", tags=["scope:work"], tags_match="all")
|
||||
call_kwargs = mock_memory.reflect_async.call_args.kwargs
|
||||
assert call_kwargs["tags"] == ["scope:work"]
|
||||
assert call_kwargs["tags_match"] == "all"
|
||||
|
||||
async def test_reflect_without_tags_no_tags_in_kwargs(self, mock_memory):
|
||||
"""When tags not provided, they should not be passed to engine."""
|
||||
mcp = _make_mcp_server(mock_memory, {"reflect"})
|
||||
await _tools(mcp)["reflect"].fn(query="test")
|
||||
call_kwargs = mock_memory.reflect_async.call_args.kwargs
|
||||
assert "tags" not in call_kwargs
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestMentalModelTrigger:
|
||||
"""Tests for trigger_refresh_after_consolidation on create/update mental model."""
|
||||
|
||||
async def test_create_with_trigger(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"create_mental_model"})
|
||||
await _tools(mcp)["create_mental_model"].fn(
|
||||
name="Test", source_query="query", trigger_refresh_after_consolidation=True
|
||||
)
|
||||
call_kwargs = mock_memory.create_mental_model.call_args.kwargs
|
||||
assert call_kwargs["trigger"] == {"refresh_after_consolidation": True}
|
||||
|
||||
async def test_create_default_trigger_false(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"create_mental_model"})
|
||||
await _tools(mcp)["create_mental_model"].fn(name="Test", source_query="query")
|
||||
call_kwargs = mock_memory.create_mental_model.call_args.kwargs
|
||||
assert call_kwargs["trigger"] == {"refresh_after_consolidation": False}
|
||||
|
||||
async def test_update_with_trigger(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"update_mental_model"})
|
||||
await _tools(mcp)["update_mental_model"].fn(
|
||||
mental_model_id="mm-1", trigger_refresh_after_consolidation=True
|
||||
)
|
||||
call_kwargs = mock_memory.update_mental_model.call_args.kwargs
|
||||
assert call_kwargs["trigger"] == {"refresh_after_consolidation": True}
|
||||
|
||||
async def test_update_without_trigger_no_trigger_in_kwargs(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"update_mental_model"})
|
||||
await _tools(mcp)["update_mental_model"].fn(mental_model_id="mm-1", name="New Name")
|
||||
call_kwargs = mock_memory.update_mental_model.call_args.kwargs
|
||||
assert "trigger" not in call_kwargs
|
||||
|
||||
|
||||
# =========================================================================
|
||||
# Directive Tool Tests
|
||||
# =========================================================================
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestDirectiveTools:
|
||||
async def test_list_directives_multi_bank(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"list_directives"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["list_directives"].fn()
|
||||
assert '"dir-1"' in result
|
||||
mock_memory.list_directives.assert_called_once()
|
||||
assert mock_memory.list_directives.call_args[0][0] == "test-bank"
|
||||
|
||||
async def test_list_directives_single_bank(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"list_directives"}, include_bank_id=False)
|
||||
result = await _tools(mcp)["list_directives"].fn()
|
||||
assert isinstance(result, dict)
|
||||
assert len(result["items"]) == 1
|
||||
|
||||
async def test_create_directive(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"create_directive"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["create_directive"].fn(name="Test", content="Be concise", priority=5)
|
||||
assert '"dir-new"' in result
|
||||
call_args = mock_memory.create_directive.call_args
|
||||
assert call_args[0][0] == "test-bank"
|
||||
assert call_args.kwargs["name"] == "Test"
|
||||
assert call_args.kwargs["content"] == "Be concise"
|
||||
assert call_args.kwargs["priority"] == 5
|
||||
|
||||
async def test_delete_directive(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"delete_directive"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["delete_directive"].fn(directive_id="dir-1")
|
||||
assert '"deleted"' in result
|
||||
assert mock_memory.delete_directive.call_args[0][1] == "dir-1"
|
||||
|
||||
async def test_delete_directive_not_found(self, mock_memory):
|
||||
mock_memory.delete_directive.return_value = False
|
||||
mcp = _make_mcp_server(mock_memory, {"delete_directive"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["delete_directive"].fn(directive_id="missing")
|
||||
assert "not found" in result
|
||||
|
||||
|
||||
# =========================================================================
|
||||
# Memory Browsing Tool Tests
|
||||
# =========================================================================
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestMemoryBrowsingTools:
|
||||
async def test_list_memories_default(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"list_memories"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["list_memories"].fn()
|
||||
assert '"mem-1"' in result
|
||||
call_kwargs = mock_memory.list_memory_units.call_args.kwargs
|
||||
assert call_kwargs["limit"] == 100
|
||||
assert call_kwargs["offset"] == 0
|
||||
|
||||
async def test_list_memories_with_filters(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"list_memories"}, include_bank_id=True)
|
||||
await _tools(mcp)["list_memories"].fn(type="world", q="test query", limit=50)
|
||||
call_kwargs = mock_memory.list_memory_units.call_args.kwargs
|
||||
assert call_kwargs["fact_type"] == "world"
|
||||
assert call_kwargs["search_query"] == "test query"
|
||||
assert call_kwargs["limit"] == 50
|
||||
|
||||
async def test_get_memory(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"get_memory"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["get_memory"].fn(memory_id="mem-1")
|
||||
assert '"mem-1"' in result
|
||||
|
||||
async def test_get_memory_not_found(self, mock_memory):
|
||||
mock_memory.get_memory_unit.return_value = None
|
||||
mcp = _make_mcp_server(mock_memory, {"get_memory"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["get_memory"].fn(memory_id="missing")
|
||||
assert "not found" in result
|
||||
|
||||
async def test_delete_memory(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"delete_memory"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["delete_memory"].fn(memory_id="mem-1")
|
||||
assert '"deleted"' in result
|
||||
assert mock_memory.delete_memory_unit.call_args.kwargs["unit_id"] == "mem-1"
|
||||
|
||||
async def test_list_memories_single_bank(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"list_memories"}, include_bank_id=False)
|
||||
result = await _tools(mcp)["list_memories"].fn()
|
||||
assert isinstance(result, dict)
|
||||
|
||||
|
||||
# =========================================================================
|
||||
# Document Tool Tests
|
||||
# =========================================================================
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestDocumentTools:
|
||||
async def test_list_documents(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"list_documents"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["list_documents"].fn()
|
||||
assert '"doc-1"' in result
|
||||
|
||||
async def test_get_document(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"get_document"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["get_document"].fn(document_id="doc-1")
|
||||
assert '"doc-1"' in result
|
||||
|
||||
async def test_get_document_not_found(self, mock_memory):
|
||||
mock_memory.get_document.return_value = None
|
||||
mcp = _make_mcp_server(mock_memory, {"get_document"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["get_document"].fn(document_id="missing")
|
||||
assert "not found" in result
|
||||
|
||||
async def test_delete_document(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"delete_document"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["delete_document"].fn(document_id="doc-1")
|
||||
assert '"deleted"' in result
|
||||
|
||||
async def test_list_documents_single_bank(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"list_documents"}, include_bank_id=False)
|
||||
result = await _tools(mcp)["list_documents"].fn()
|
||||
assert isinstance(result, dict)
|
||||
|
||||
|
||||
# =========================================================================
|
||||
# Operation Tool Tests
|
||||
# =========================================================================
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestOperationTools:
|
||||
async def test_list_operations(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"list_operations"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["list_operations"].fn()
|
||||
assert '"op-1"' in result
|
||||
|
||||
async def test_list_operations_with_status(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"list_operations"}, include_bank_id=True)
|
||||
await _tools(mcp)["list_operations"].fn(status="completed", limit=10)
|
||||
call_kwargs = mock_memory.list_operations.call_args.kwargs
|
||||
assert call_kwargs["status"] == "completed"
|
||||
assert call_kwargs["limit"] == 10
|
||||
|
||||
async def test_get_operation(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"get_operation"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["get_operation"].fn(operation_id="op-1")
|
||||
assert '"op-1"' in result
|
||||
|
||||
async def test_cancel_operation(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"cancel_operation"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["cancel_operation"].fn(operation_id="op-1")
|
||||
assert '"cancelled"' in result
|
||||
|
||||
async def test_list_operations_single_bank(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"list_operations"}, include_bank_id=False)
|
||||
result = await _tools(mcp)["list_operations"].fn()
|
||||
assert isinstance(result, dict)
|
||||
|
||||
|
||||
# =========================================================================
|
||||
# Tags & Bank Tool Tests
|
||||
# =========================================================================
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestTagsAndBankTools:
|
||||
async def test_list_tags(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"list_tags"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["list_tags"].fn(q="project:*", limit=50)
|
||||
call_kwargs = mock_memory.list_tags.call_args.kwargs
|
||||
assert call_kwargs["pattern"] == "project:*"
|
||||
assert call_kwargs["limit"] == 50
|
||||
|
||||
async def test_get_bank(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"get_bank"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["get_bank"].fn()
|
||||
assert '"test-bank"' in result or "test-bank" in result
|
||||
|
||||
async def test_get_bank_stats(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"get_bank_stats"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["get_bank_stats"].fn()
|
||||
assert "100" in result # nodes count
|
||||
|
||||
async def test_update_bank(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"update_bank"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["update_bank"].fn(name="New Name", mission="New Mission")
|
||||
call_kwargs = mock_memory.update_bank.call_args.kwargs
|
||||
assert call_kwargs["name"] == "New Name"
|
||||
assert call_kwargs["mission"] == "New Mission"
|
||||
|
||||
async def test_delete_bank(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"delete_bank"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["delete_bank"].fn()
|
||||
assert '"deleted"' in result
|
||||
mock_memory.delete_bank.assert_called_once()
|
||||
|
||||
async def test_clear_memories(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"clear_memories"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["clear_memories"].fn()
|
||||
assert '"cleared"' in result
|
||||
mock_memory.delete_bank.assert_called_once()
|
||||
|
||||
async def test_clear_memories_with_type_filter(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"clear_memories"}, include_bank_id=True)
|
||||
await _tools(mcp)["clear_memories"].fn(type="world")
|
||||
call_kwargs = mock_memory.delete_bank.call_args.kwargs
|
||||
assert call_kwargs["fact_type"] == "world"
|
||||
|
||||
async def test_list_tags_single_bank(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"list_tags"}, include_bank_id=False)
|
||||
result = await _tools(mcp)["list_tags"].fn()
|
||||
assert isinstance(result, dict)
|
||||
|
||||
async def test_get_bank_single_bank(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"get_bank"}, include_bank_id=False)
|
||||
result = await _tools(mcp)["get_bank"].fn()
|
||||
assert isinstance(result, dict)
|
||||
|
||||
async def test_delete_bank_single_bank(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"delete_bank"}, include_bank_id=False)
|
||||
result = await _tools(mcp)["delete_bank"].fn()
|
||||
assert isinstance(result, dict)
|
||||
assert result["status"] == "deleted"
|
||||
|
||||
async def test_clear_memories_single_bank(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"clear_memories"}, include_bank_id=False)
|
||||
result = await _tools(mcp)["clear_memories"].fn()
|
||||
assert isinstance(result, dict)
|
||||
assert result["status"] == "cleared"
|
||||
|
||||
|
||||
# =========================================================================
|
||||
# Additional Error Handling & Edge Case Tests
|
||||
# =========================================================================
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestOperationErrorHandling:
|
||||
"""Error handling tests for operation tools."""
|
||||
|
||||
async def test_get_operation_engine_error(self, mock_memory):
|
||||
mock_memory.get_operation_status.side_effect = RuntimeError("Operation not found")
|
||||
mcp = _make_mcp_server(mock_memory, {"get_operation"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["get_operation"].fn(operation_id="missing")
|
||||
assert "error" in result
|
||||
assert "Operation not found" in result
|
||||
|
||||
async def test_get_operation_engine_error_single_bank(self, mock_memory):
|
||||
mock_memory.get_operation_status.side_effect = RuntimeError("Operation not found")
|
||||
mcp = _make_mcp_server(mock_memory, {"get_operation"}, include_bank_id=False)
|
||||
result = await _tools(mcp)["get_operation"].fn(operation_id="missing")
|
||||
assert isinstance(result, dict)
|
||||
assert "Operation not found" in result["error"]
|
||||
|
||||
async def test_cancel_operation_engine_error(self, mock_memory):
|
||||
mock_memory.cancel_operation.side_effect = RuntimeError("Cannot cancel completed operation")
|
||||
mcp = _make_mcp_server(mock_memory, {"cancel_operation"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["cancel_operation"].fn(operation_id="op-done")
|
||||
assert "error" in result
|
||||
assert "Cannot cancel" in result
|
||||
|
||||
async def test_cancel_operation_engine_error_single_bank(self, mock_memory):
|
||||
mock_memory.cancel_operation.side_effect = RuntimeError("Cannot cancel")
|
||||
mcp = _make_mcp_server(mock_memory, {"cancel_operation"}, include_bank_id=False)
|
||||
result = await _tools(mcp)["cancel_operation"].fn(operation_id="op-done")
|
||||
assert isinstance(result, dict)
|
||||
assert "Cannot cancel" in result["error"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestDeleteErrorHandling:
|
||||
"""Error handling tests for delete operations."""
|
||||
|
||||
async def test_delete_memory_engine_error(self, mock_memory):
|
||||
mock_memory.delete_memory_unit.side_effect = RuntimeError("DB error")
|
||||
mcp = _make_mcp_server(mock_memory, {"delete_memory"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["delete_memory"].fn(memory_id="mem-1")
|
||||
assert "error" in result
|
||||
assert "DB error" in result
|
||||
|
||||
async def test_delete_memory_single_bank(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"delete_memory"}, include_bank_id=False)
|
||||
result = await _tools(mcp)["delete_memory"].fn(memory_id="mem-1")
|
||||
assert isinstance(result, dict)
|
||||
assert result["status"] == "deleted"
|
||||
|
||||
async def test_delete_document_engine_error(self, mock_memory):
|
||||
mock_memory.delete_document.side_effect = RuntimeError("DB error")
|
||||
mcp = _make_mcp_server(mock_memory, {"delete_document"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["delete_document"].fn(document_id="doc-1")
|
||||
assert "error" in result
|
||||
assert "DB error" in result
|
||||
|
||||
async def test_delete_document_single_bank(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"delete_document"}, include_bank_id=False)
|
||||
result = await _tools(mcp)["delete_document"].fn(document_id="doc-1")
|
||||
assert isinstance(result, dict)
|
||||
assert result["status"] == "deleted"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestUpdateBankVariants:
|
||||
"""Additional tests for update_bank tool."""
|
||||
|
||||
async def test_update_bank_single_bank(self, mock_memory):
|
||||
mcp = _make_mcp_server(mock_memory, {"update_bank"}, include_bank_id=False)
|
||||
result = await _tools(mcp)["update_bank"].fn(name="New Name")
|
||||
assert isinstance(result, dict)
|
||||
call_kwargs = mock_memory.update_bank.call_args.kwargs
|
||||
assert call_kwargs["name"] == "New Name"
|
||||
|
||||
async def test_update_bank_engine_error(self, mock_memory):
|
||||
mock_memory.update_bank.side_effect = RuntimeError("DB error")
|
||||
mcp = _make_mcp_server(mock_memory, {"update_bank"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["update_bank"].fn(name="X")
|
||||
assert "error" in result
|
||||
|
||||
async def test_get_bank_stats_engine_error(self, mock_memory):
|
||||
mock_memory.get_bank_stats.side_effect = RuntimeError("DB error")
|
||||
mcp = _make_mcp_server(mock_memory, {"get_bank_stats"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["get_bank_stats"].fn()
|
||||
assert "error" in result
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestEmptyListReturns:
|
||||
"""Tests that empty lists are handled gracefully."""
|
||||
|
||||
async def test_list_memories_empty(self, mock_memory):
|
||||
mock_memory.list_memory_units.return_value = {"items": [], "total": 0}
|
||||
mcp = _make_mcp_server(mock_memory, {"list_memories"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["list_memories"].fn()
|
||||
assert '"items": []' in result or "[]" in result
|
||||
|
||||
async def test_list_documents_empty(self, mock_memory):
|
||||
mock_memory.list_documents.return_value = {"items": [], "total": 0}
|
||||
mcp = _make_mcp_server(mock_memory, {"list_documents"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["list_documents"].fn()
|
||||
assert '"items": []' in result or "[]" in result
|
||||
|
||||
async def test_list_operations_empty(self, mock_memory):
|
||||
mock_memory.list_operations.return_value = {"items": []}
|
||||
mcp = _make_mcp_server(mock_memory, {"list_operations"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["list_operations"].fn()
|
||||
assert '"items": []' in result or "[]" in result
|
||||
|
||||
async def test_list_directives_empty(self, mock_memory):
|
||||
mock_memory.list_directives.return_value = []
|
||||
mcp = _make_mcp_server(mock_memory, {"list_directives"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["list_directives"].fn()
|
||||
assert "[]" in result
|
||||
|
||||
async def test_list_tags_empty(self, mock_memory):
|
||||
mock_memory.list_tags.return_value = {"items": [], "total": 0}
|
||||
mcp = _make_mcp_server(mock_memory, {"list_tags"}, include_bank_id=True)
|
||||
result = await _tools(mcp)["list_tags"].fn()
|
||||
assert '"items": []' in result or "[]" in result
|
||||
|
||||
# =========================================================================
|
||||
# Bank-Level Tool Filtering Tests
|
||||
# =========================================================================
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_memory_with_resolver():
|
||||
"""Create a mock MemoryEngine with config resolver for bank filtering tests."""
|
||||
memory = MagicMock()
|
||||
memory.retain_batch_async = AsyncMock()
|
||||
memory.recall_async = AsyncMock(
|
||||
return_value=MagicMock(
|
||||
model_dump_json=lambda indent=None: '{"results": []}',
|
||||
model_dump=lambda: {"results": []},
|
||||
)
|
||||
)
|
||||
memory._config_resolver = MagicMock()
|
||||
memory._config_resolver.get_bank_config = AsyncMock(return_value={})
|
||||
return memory
|
||||
|
||||
|
||||
class TestBankToolFiltering:
|
||||
"""Tests for bank-level mcp_enabled_tools filtering via _apply_bank_tool_filtering."""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_disallowed_tool_raises_error(self, mock_memory_with_resolver):
|
||||
"""Tool not in bank's mcp_enabled_tools list is hidden from get_tools()."""
|
||||
from fastmcp import FastMCP
|
||||
|
||||
mock_memory_with_resolver._config_resolver.get_bank_config = AsyncMock(
|
||||
return_value={"mcp_enabled_tools": ["retain"]}
|
||||
)
|
||||
|
||||
mcp = FastMCP("test")
|
||||
config = MCPToolsConfig(
|
||||
bank_id_resolver=lambda: "test-bank",
|
||||
include_bank_id_param=False,
|
||||
tools={"retain", "recall"},
|
||||
)
|
||||
register_mcp_tools(mcp, mock_memory_with_resolver, config)
|
||||
|
||||
# Both tools are registered in the manager's internal dict
|
||||
assert "recall" in mcp._tool_manager._tools
|
||||
|
||||
# But get_tools() (used by tools/list and tools/call) filters it out
|
||||
visible = await mcp._tool_manager.get_tools()
|
||||
assert "retain" in visible
|
||||
assert "recall" not in visible
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_allowed_tool_remains_visible(self, mock_memory_with_resolver):
|
||||
"""Tool in bank's mcp_enabled_tools list stays visible in get_tools()."""
|
||||
from fastmcp import FastMCP
|
||||
|
||||
mock_memory_with_resolver._config_resolver.get_bank_config = AsyncMock(
|
||||
return_value={"mcp_enabled_tools": ["retain", "recall"]}
|
||||
)
|
||||
|
||||
mcp = FastMCP("test")
|
||||
config = MCPToolsConfig(
|
||||
bank_id_resolver=lambda: "test-bank",
|
||||
include_bank_id_param=False,
|
||||
tools={"retain", "recall"},
|
||||
)
|
||||
register_mcp_tools(mcp, mock_memory_with_resolver, config)
|
||||
|
||||
visible = await mcp._tool_manager.get_tools()
|
||||
assert "retain" in visible
|
||||
assert "recall" in visible
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_no_filter_when_mcp_enabled_tools_absent(self, mock_memory_with_resolver):
|
||||
"""When bank config has no mcp_enabled_tools key, all tools remain visible."""
|
||||
from fastmcp import FastMCP
|
||||
|
||||
mock_memory_with_resolver._config_resolver.get_bank_config = AsyncMock(return_value={})
|
||||
|
||||
mcp = FastMCP("test")
|
||||
config = MCPToolsConfig(
|
||||
bank_id_resolver=lambda: "test-bank",
|
||||
include_bank_id_param=False,
|
||||
tools={"retain", "recall"},
|
||||
)
|
||||
register_mcp_tools(mcp, mock_memory_with_resolver, config)
|
||||
|
||||
visible = await mcp._tool_manager.get_tools()
|
||||
assert "retain" in visible
|
||||
assert "recall" in visible
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_filter_skipped_when_no_bank_id(self, mock_memory_with_resolver):
|
||||
"""When bank_id resolver returns None, config is not fetched and all tools are visible."""
|
||||
from fastmcp import FastMCP
|
||||
|
||||
mock_memory_with_resolver._config_resolver.get_bank_config = AsyncMock(
|
||||
return_value={"mcp_enabled_tools": ["retain"]} # Would block recall
|
||||
)
|
||||
|
||||
mcp = FastMCP("test")
|
||||
config = MCPToolsConfig(
|
||||
bank_id_resolver=lambda: None, # No bank_id context
|
||||
include_bank_id_param=False,
|
||||
tools={"retain", "recall"},
|
||||
)
|
||||
register_mcp_tools(mcp, mock_memory_with_resolver, config)
|
||||
|
||||
visible = await mcp._tool_manager.get_tools()
|
||||
# Filter bypassed — config resolver was never consulted, all tools visible
|
||||
assert "recall" in visible
|
||||
mock_memory_with_resolver._config_resolver.get_bank_config.assert_not_called()
|
||||
|
||||
@@ -404,25 +404,12 @@ class TestDirectivesInReflect:
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
# Run reflect query
|
||||
result = await memory.reflect_async(
|
||||
bank_id=bank_id,
|
||||
query="What does Alice do for work?",
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
assert result.text is not None
|
||||
assert len(result.text) > 0
|
||||
|
||||
# Check that the response contains French words/patterns
|
||||
# Common French words that would appear when talking about someone's job
|
||||
french_indicators = [
|
||||
"elle",
|
||||
"travaille",
|
||||
"est",
|
||||
"une",
|
||||
"le",
|
||||
"la",
|
||||
"qui",
|
||||
"chez",
|
||||
"logiciel",
|
||||
@@ -430,11 +417,27 @@ class TestDirectivesInReflect:
|
||||
"ingénieure",
|
||||
"développeur",
|
||||
"développeuse",
|
||||
"ingénierie",
|
||||
"française",
|
||||
]
|
||||
response_lower = result.text.lower()
|
||||
|
||||
# At least some French words should appear in the response
|
||||
french_word_count = sum(1 for word in french_indicators if word in response_lower)
|
||||
# Run reflect query (retry once since small LLMs may not always follow language directives)
|
||||
french_word_count = 0
|
||||
for _attempt in range(2):
|
||||
result = await memory.reflect_async(
|
||||
bank_id=bank_id,
|
||||
query="What does Alice do for work?",
|
||||
request_context=request_context,
|
||||
)
|
||||
assert result.text is not None
|
||||
assert len(result.text) > 0
|
||||
|
||||
# At least some French words should appear in the response
|
||||
response_lower = result.text.lower()
|
||||
french_word_count = sum(1 for word in french_indicators if word in response_lower)
|
||||
if french_word_count >= 2:
|
||||
break
|
||||
|
||||
assert (
|
||||
french_word_count >= 2
|
||||
), f"Expected French response, but got: {result.text[:200]}"
|
||||
@@ -474,7 +477,7 @@ class TestDirectivesInReflect:
|
||||
await memory.create_directive(
|
||||
bank_id=bank_id,
|
||||
name="General Policy",
|
||||
content="Always be polite and start responses with 'Hello!'",
|
||||
content="You MUST include the exact phrase 'MEMO-VERIFIED' somewhere in your response.",
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
@@ -482,7 +485,7 @@ class TestDirectivesInReflect:
|
||||
await memory.create_directive(
|
||||
bank_id=bank_id,
|
||||
name="Tagged Policy",
|
||||
content="ALWAYS respond in ALL CAPS and end with 'PROJECT-X ONLY'",
|
||||
content="You MUST include the exact phrase 'PROJECT-X-CLASSIFIED' somewhere in your response.",
|
||||
tags=["project-x"],
|
||||
request_context=request_context,
|
||||
)
|
||||
@@ -494,18 +497,16 @@ class TestDirectivesInReflect:
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
response_lower = result.text.lower()
|
||||
# Verify the isolation mechanism: only untagged directive should be loaded
|
||||
untagged_directive_names = [d.name for d in result.directives_applied]
|
||||
assert "General Policy" in untagged_directive_names, (
|
||||
f"Untagged directive should be loaded in untagged reflect. Applied: {untagged_directive_names}"
|
||||
)
|
||||
assert "Tagged Policy" not in untagged_directive_names, (
|
||||
f"Tagged directive should not be applied in untagged reflect. Applied: {untagged_directive_names}"
|
||||
)
|
||||
|
||||
# Should follow the untagged directive (polite greeting)
|
||||
assert "hello" in response_lower, f"Expected 'Hello' from untagged directive, but got: {result.text}"
|
||||
|
||||
# Should NOT follow the tagged directive (all caps and PROJECT-X)
|
||||
# If it did follow, the entire response would be in caps
|
||||
all_caps = result.text.replace(" ", "").replace("!", "").replace(".", "").isupper()
|
||||
assert not all_caps, f"Tagged directive was incorrectly applied to untagged operation: {result.text}"
|
||||
assert "project-x only" not in response_lower, f"Tagged directive was incorrectly applied: {result.text}"
|
||||
|
||||
# Now run reflect WITH the tag - should apply BOTH directives
|
||||
# Now run reflect WITH the tag - should load BOTH directives
|
||||
result_tagged = await memory.reflect_async(
|
||||
bank_id=bank_id,
|
||||
query="What color is the sky?",
|
||||
@@ -514,10 +515,14 @@ class TestDirectivesInReflect:
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
response_tagged_lower = result_tagged.text.lower()
|
||||
|
||||
# With strict matching and tags, should apply the tagged directive
|
||||
assert "project-x only" in response_tagged_lower, f"Tagged directive should be applied with tags: {result_tagged.text}"
|
||||
# Verify the isolation mechanism: both directives should be loaded when tags match
|
||||
tagged_directive_names = [d.name for d in result_tagged.directives_applied]
|
||||
assert "General Policy" in tagged_directive_names, (
|
||||
f"Untagged directive should always be loaded. Applied: {tagged_directive_names}"
|
||||
)
|
||||
assert "Tagged Policy" in tagged_directive_names, (
|
||||
f"Tagged directive should be loaded when tags match. Applied: {tagged_directive_names}"
|
||||
)
|
||||
|
||||
# Cleanup
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
@@ -15,6 +15,10 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.xfail(
|
||||
strict=False,
|
||||
reason="Gemini sometimes consistently translates Chinese content to English despite instructions",
|
||||
)
|
||||
async def test_retain_chinese_content(memory, request_context):
|
||||
"""
|
||||
Test that retain correctly extracts facts from Chinese content
|
||||
@@ -24,70 +28,87 @@ async def test_retain_chinese_content(memory, request_context):
|
||||
1. Facts are extracted from Chinese text
|
||||
2. The extracted facts contain Chinese characters
|
||||
3. Entity names are preserved in Chinese
|
||||
|
||||
Note: LLM fact extraction is non-deterministic and may sometimes translate
|
||||
content to English despite instructions. We retry up to 3 times.
|
||||
"""
|
||||
bank_id = f"test_chinese_retain_{datetime.now(timezone.utc).timestamp()}"
|
||||
max_retries = 3
|
||||
last_error = None
|
||||
|
||||
try:
|
||||
# Chinese content about a person and their activities
|
||||
chinese_content = """
|
||||
张伟是一位资深软件工程师,在腾讯工作了五年。他专门研究分布式系统,
|
||||
并领导了公司微服务架构的开发。他以编写干净、文档完善的代码而闻名。
|
||||
for attempt in range(max_retries):
|
||||
bank_id = f"test_chinese_retain_{datetime.now(timezone.utc).timestamp()}_{attempt}"
|
||||
|
||||
李明上个月加入团队担任初级开发人员。他正在学习React和Node.js。
|
||||
李明很有热情,在代码审查中提出很好的问题。他最近完成了他的第一个功能,
|
||||
这是一个用户认证流程。
|
||||
try:
|
||||
# Chinese content about a person and their activities
|
||||
chinese_content = """
|
||||
张伟是一位资深软件工程师,在腾讯工作了五年。他专门研究分布式系统,
|
||||
并领导了公司微服务架构的开发。他以编写干净、文档完善的代码而闻名。
|
||||
|
||||
团队使用Kubernetes进行容器编排,并部署到阿里云。他们遵循敏捷方法论,
|
||||
采用两周冲刺周期。合并前必须进行代码审查。
|
||||
"""
|
||||
李明上个月加入团队担任初级开发人员。他正在学习React和Node.js。
|
||||
李明很有热情,在代码审查中提出很好的问题。他最近完成了他的第一个功能,
|
||||
这是一个用户认证流程。
|
||||
|
||||
# Retain the Chinese content
|
||||
unit_ids = await memory.retain_async(
|
||||
bank_id=bank_id,
|
||||
content=chinese_content,
|
||||
context="团队概述", # Chinese context
|
||||
event_date=datetime(2024, 1, 15, tzinfo=timezone.utc),
|
||||
request_context=request_context,
|
||||
)
|
||||
团队使用Kubernetes进行容器编排,并部署到阿里云。他们遵循敏捷方法论,
|
||||
采用两周冲刺周期。合并前必须进行代码审查。
|
||||
"""
|
||||
|
||||
logger.info(f"Retained {len(unit_ids)} facts from Chinese content")
|
||||
assert len(unit_ids) > 0, "Should have extracted and stored facts from Chinese content"
|
||||
# Retain the Chinese content
|
||||
unit_ids = await memory.retain_async(
|
||||
bank_id=bank_id,
|
||||
content=chinese_content,
|
||||
context="团队概述", # Chinese context
|
||||
event_date=datetime(2024, 1, 15, tzinfo=timezone.utc),
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
# Recall the facts with a Chinese query
|
||||
result = await memory.recall_async(
|
||||
bank_id=bank_id,
|
||||
query="告诉我关于张伟的信息", # "Tell me about Zhang Wei"
|
||||
budget=Budget.MID,
|
||||
max_tokens=1000,
|
||||
fact_type=["world"],
|
||||
request_context=request_context,
|
||||
)
|
||||
logger.info(f"Retained {len(unit_ids)} facts from Chinese content (attempt {attempt + 1})")
|
||||
assert len(unit_ids) > 0, "Should have extracted and stored facts from Chinese content"
|
||||
|
||||
logger.info(f"Recalled {len(result.results)} facts")
|
||||
assert len(result.results) > 0, "Should recall facts about Zhang Wei"
|
||||
# Recall the facts with a Chinese query
|
||||
result = await memory.recall_async(
|
||||
bank_id=bank_id,
|
||||
query="告诉我关于张伟的信息", # "Tell me about Zhang Wei"
|
||||
budget=Budget.MID,
|
||||
max_tokens=1000,
|
||||
fact_type=["world"],
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
# Verify that the facts contain Chinese characters
|
||||
# At least one fact should mention 张伟 (Zhang Wei) or related Chinese content
|
||||
chinese_facts_found = 0
|
||||
for fact in result.results:
|
||||
logger.info(f"Fact: {fact.text[:100]}...")
|
||||
# Check for common Chinese characters or the name
|
||||
if any(
|
||||
char in fact.text
|
||||
for char in ["张", "伟", "腾讯", "软件", "工程师", "分布式", "系统", "代码"]
|
||||
):
|
||||
chinese_facts_found += 1
|
||||
logger.info(f"Recalled {len(result.results)} facts")
|
||||
assert len(result.results) > 0, "Should recall facts about Zhang Wei"
|
||||
|
||||
logger.info(f"Found {chinese_facts_found} facts with Chinese content")
|
||||
assert chinese_facts_found > 0, (
|
||||
f"Expected facts to contain Chinese characters, but none found. "
|
||||
f"Facts: {[f.text for f in result.results]}"
|
||||
)
|
||||
# Verify that the facts contain Chinese characters
|
||||
# At least one fact should mention 张伟 (Zhang Wei) or related Chinese content
|
||||
chinese_facts_found = 0
|
||||
for fact in result.results:
|
||||
logger.info(f"Fact: {fact.text[:100]}...")
|
||||
# Check for common Chinese characters or the name
|
||||
if any(
|
||||
char in fact.text
|
||||
for char in ["张", "伟", "腾讯", "软件", "工程师", "分布式", "系统", "代码"]
|
||||
):
|
||||
chinese_facts_found += 1
|
||||
|
||||
logger.info("Chinese retain test passed - facts preserved in Chinese")
|
||||
logger.info(f"Found {chinese_facts_found} facts with Chinese content")
|
||||
assert chinese_facts_found > 0, (
|
||||
f"Expected facts to contain Chinese characters, but none found. "
|
||||
f"Facts: {[f.text for f in result.results]}"
|
||||
)
|
||||
|
||||
finally:
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
logger.info("Chinese retain test passed - facts preserved in Chinese")
|
||||
return # Test passed
|
||||
|
||||
except AssertionError as e:
|
||||
last_error = e
|
||||
if attempt < max_retries - 1:
|
||||
logger.warning(f"Attempt {attempt + 1} failed: {e}. Retrying...")
|
||||
else:
|
||||
raise e
|
||||
finally:
|
||||
try:
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@@ -133,7 +154,7 @@ async def test_reflect_chinese_content(memory, request_context):
|
||||
result = await memory.reflect_async(
|
||||
bank_id=bank_id,
|
||||
query=query,
|
||||
budget=Budget.LOW,
|
||||
budget=Budget.MID,
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
|
||||
@@ -0,0 +1,457 @@
|
||||
"""
|
||||
Tests for observation invalidation when source memories are deleted.
|
||||
|
||||
These tests verify that:
|
||||
1. Observations are deleted (not just updated) when their source memories are removed
|
||||
2. Remaining source memories are reset for re-consolidation (consolidated_at=NULL)
|
||||
3. The clear_observations_for_memory method correctly clears observations and
|
||||
resets the target memory itself for re-consolidation
|
||||
4. delete_bank(fact_type=...) also cleans up affected observations
|
||||
"""
|
||||
import uuid
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from hindsight_api import RequestContext
|
||||
from hindsight_api.engine.memory_engine import MemoryEngine
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
async def _insert_memory(conn, bank_id: str, text: str, fact_type: str = "experience") -> uuid.UUID:
|
||||
"""Insert a memory unit directly, bypassing LLM retain pipeline."""
|
||||
mem_id = uuid.uuid4()
|
||||
await conn.execute(
|
||||
"""
|
||||
INSERT INTO memory_units (id, bank_id, text, fact_type, event_date, created_at, updated_at, consolidated_at)
|
||||
VALUES ($1, $2, $3, $4, NOW(), NOW(), NOW(), NOW())
|
||||
""",
|
||||
mem_id,
|
||||
bank_id,
|
||||
text,
|
||||
fact_type,
|
||||
)
|
||||
return mem_id
|
||||
|
||||
|
||||
async def _insert_observation(
|
||||
conn, bank_id: str, text: str, source_memory_ids: list[uuid.UUID]
|
||||
) -> uuid.UUID:
|
||||
"""Insert an observation unit directly."""
|
||||
obs_id = uuid.uuid4()
|
||||
await conn.execute(
|
||||
"""
|
||||
INSERT INTO memory_units (
|
||||
id, bank_id, text, fact_type, event_date, source_memory_ids, proof_count, created_at, updated_at
|
||||
) VALUES ($1, $2, $3, 'observation', NOW(), $4, $5, NOW(), NOW())
|
||||
""",
|
||||
obs_id,
|
||||
bank_id,
|
||||
text,
|
||||
source_memory_ids,
|
||||
len(source_memory_ids),
|
||||
)
|
||||
return obs_id
|
||||
|
||||
|
||||
async def _get_observation_ids(conn, bank_id: str) -> list[str]:
|
||||
rows = await conn.fetch(
|
||||
"SELECT id FROM memory_units WHERE bank_id = $1 AND fact_type = 'observation'",
|
||||
bank_id,
|
||||
)
|
||||
return [str(r["id"]) for r in rows]
|
||||
|
||||
|
||||
async def _get_consolidated_at(conn, memory_id: uuid.UUID):
|
||||
return await conn.fetchval(
|
||||
"SELECT consolidated_at FROM memory_units WHERE id = $1",
|
||||
memory_id,
|
||||
)
|
||||
|
||||
|
||||
async def _ensure_bank(memory: MemoryEngine, bank_id: str, request_context: RequestContext):
|
||||
await memory.get_bank_profile(bank_id=bank_id, request_context=request_context)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests: delete_memory_unit
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestDeleteMemoryUnitObservationCleanup:
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_deleting_source_memory_removes_observation(
|
||||
self, memory: MemoryEngine, request_context: RequestContext
|
||||
):
|
||||
"""Deleting a source memory removes observations derived from it."""
|
||||
bank_id = f"test-invalidate-del-{uuid.uuid4().hex[:8]}"
|
||||
await _ensure_bank(memory, bank_id, request_context)
|
||||
|
||||
pool = await memory._get_pool()
|
||||
async with pool.acquire() as conn:
|
||||
m1 = await _insert_memory(conn, bank_id, "Alice loves hiking.")
|
||||
m2 = await _insert_memory(conn, bank_id, "Alice goes hiking every weekend.")
|
||||
obs_id = await _insert_observation(conn, bank_id, "Alice enjoys hiking regularly.", [m1, m2])
|
||||
|
||||
await memory.delete_memory_unit(str(m1), request_context=request_context)
|
||||
|
||||
async with pool.acquire() as conn:
|
||||
obs_ids = await _get_observation_ids(conn, bank_id)
|
||||
assert str(obs_id) not in obs_ids, "Observation should have been deleted"
|
||||
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_deleting_source_memory_resets_remaining_source_consolidated_at(
|
||||
self, memory: MemoryEngine, request_context: RequestContext
|
||||
):
|
||||
"""After deleting a source memory, remaining source memories are reset for re-consolidation."""
|
||||
bank_id = f"test-invalidate-reset-{uuid.uuid4().hex[:8]}"
|
||||
await _ensure_bank(memory, bank_id, request_context)
|
||||
|
||||
pool = await memory._get_pool()
|
||||
async with pool.acquire() as conn:
|
||||
m1 = await _insert_memory(conn, bank_id, "Alice loves hiking.")
|
||||
m2 = await _insert_memory(conn, bank_id, "Alice goes hiking every weekend.")
|
||||
await _insert_observation(conn, bank_id, "Alice enjoys hiking regularly.", [m1, m2])
|
||||
|
||||
# Verify m2 starts with consolidated_at set
|
||||
assert await _get_consolidated_at(conn, m2) is not None
|
||||
|
||||
# Patch out consolidation so it doesn't re-set consolidated_at before we can check it
|
||||
with patch.object(memory, "submit_async_consolidation", new=AsyncMock()):
|
||||
await memory.delete_memory_unit(str(m1), request_context=request_context)
|
||||
|
||||
async with pool.acquire() as conn:
|
||||
# m2 should have consolidated_at reset to NULL
|
||||
consolidated_at = await _get_consolidated_at(conn, m2)
|
||||
assert consolidated_at is None, "Remaining source memory should be reset for re-consolidation"
|
||||
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_deleting_non_source_memory_leaves_observations_intact(
|
||||
self, memory: MemoryEngine, request_context: RequestContext
|
||||
):
|
||||
"""Deleting a memory that is not a source of any observation leaves observations unchanged."""
|
||||
bank_id = f"test-invalidate-noop-{uuid.uuid4().hex[:8]}"
|
||||
await _ensure_bank(memory, bank_id, request_context)
|
||||
|
||||
pool = await memory._get_pool()
|
||||
async with pool.acquire() as conn:
|
||||
m1 = await _insert_memory(conn, bank_id, "Alice loves hiking.")
|
||||
m2 = await _insert_memory(conn, bank_id, "Alice goes hiking every weekend.")
|
||||
unrelated = await _insert_memory(conn, bank_id, "Bob likes cycling.")
|
||||
obs_id = await _insert_observation(conn, bank_id, "Alice enjoys hiking regularly.", [m1, m2])
|
||||
|
||||
await memory.delete_memory_unit(str(unrelated), request_context=request_context)
|
||||
|
||||
async with pool.acquire() as conn:
|
||||
obs_ids = await _get_observation_ids(conn, bank_id)
|
||||
assert str(obs_id) in obs_ids, "Observation should remain untouched"
|
||||
# m1 and m2 should still be consolidated
|
||||
assert await _get_consolidated_at(conn, m1) is not None
|
||||
assert await _get_consolidated_at(conn, m2) is not None
|
||||
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_deleting_sole_source_memory_removes_observation_no_remaining_reset(
|
||||
self, memory: MemoryEngine, request_context: RequestContext
|
||||
):
|
||||
"""When an observation has only one source and it's deleted, observation is removed with no remaining memories to reset."""
|
||||
bank_id = f"test-invalidate-sole-{uuid.uuid4().hex[:8]}"
|
||||
await _ensure_bank(memory, bank_id, request_context)
|
||||
|
||||
pool = await memory._get_pool()
|
||||
async with pool.acquire() as conn:
|
||||
m1 = await _insert_memory(conn, bank_id, "Alice loves hiking.")
|
||||
obs_id = await _insert_observation(conn, bank_id, "Alice enjoys hiking.", [m1])
|
||||
|
||||
await memory.delete_memory_unit(str(m1), request_context=request_context)
|
||||
|
||||
async with pool.acquire() as conn:
|
||||
obs_ids = await _get_observation_ids(conn, bank_id)
|
||||
assert str(obs_id) not in obs_ids, "Observation should have been deleted"
|
||||
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_deleting_observation_type_memory_does_not_trigger_invalidation(
|
||||
self, memory: MemoryEngine, request_context: RequestContext
|
||||
):
|
||||
"""Deleting a memory with fact_type='observation' directly does not trigger invalidation logic."""
|
||||
bank_id = f"test-invalidate-obstype-{uuid.uuid4().hex[:8]}"
|
||||
await _ensure_bank(memory, bank_id, request_context)
|
||||
|
||||
pool = await memory._get_pool()
|
||||
async with pool.acquire() as conn:
|
||||
m1 = await _insert_memory(conn, bank_id, "Alice loves hiking.")
|
||||
obs_id = await _insert_observation(conn, bank_id, "Alice enjoys hiking.", [m1])
|
||||
|
||||
# Delete the observation directly (not the source memory)
|
||||
await memory.delete_memory_unit(str(obs_id), request_context=request_context)
|
||||
|
||||
async with pool.acquire() as conn:
|
||||
# Source memory should still be consolidated (not reset)
|
||||
assert await _get_consolidated_at(conn, m1) is not None
|
||||
obs_ids = await _get_observation_ids(conn, bank_id)
|
||||
assert str(obs_id) not in obs_ids
|
||||
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests: delete_document
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestDeleteDocumentObservationCleanup:
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_deleting_document_removes_observations(
|
||||
self, memory: MemoryEngine, request_context: RequestContext
|
||||
):
|
||||
"""Deleting a document removes observations derived from its memory units."""
|
||||
bank_id = f"test-invalidate-doc-{uuid.uuid4().hex[:8]}"
|
||||
await _ensure_bank(memory, bank_id, request_context)
|
||||
|
||||
pool = await memory._get_pool()
|
||||
|
||||
# Create a document and attach memories to it
|
||||
async with pool.acquire() as conn:
|
||||
doc_id = str(uuid.uuid4()) # documents.id is TEXT
|
||||
await conn.execute(
|
||||
"""
|
||||
INSERT INTO documents (id, bank_id, original_text, content_hash, created_at, updated_at)
|
||||
VALUES ($1, $2, 'some doc', 'hash123', NOW(), NOW())
|
||||
""",
|
||||
doc_id,
|
||||
bank_id,
|
||||
)
|
||||
m1 = uuid.uuid4()
|
||||
m2 = uuid.uuid4()
|
||||
for mem_id, text in [(m1, "Alice loves hiking."), (m2, "Alice goes hiking every weekend.")]:
|
||||
await conn.execute(
|
||||
"""
|
||||
INSERT INTO memory_units (id, bank_id, text, fact_type, event_date, document_id, created_at, updated_at, consolidated_at)
|
||||
VALUES ($1, $2, $3, 'experience', NOW(), $4, NOW(), NOW(), NOW())
|
||||
""",
|
||||
mem_id,
|
||||
bank_id,
|
||||
text,
|
||||
doc_id,
|
||||
)
|
||||
|
||||
# Standalone memory (not in document)
|
||||
m3 = await _insert_memory(conn, bank_id, "Alice is an avid outdoor person.")
|
||||
|
||||
# Observation referencing both doc memories and the standalone memory
|
||||
obs_id = await _insert_observation(
|
||||
conn, bank_id, "Alice enjoys outdoor activities.", [m1, m2, m3]
|
||||
)
|
||||
|
||||
# Patch out consolidation so it doesn't re-set consolidated_at before we can check it
|
||||
with patch.object(memory, "submit_async_consolidation", new=AsyncMock()):
|
||||
await memory.delete_document(str(doc_id), bank_id, request_context=request_context)
|
||||
|
||||
async with pool.acquire() as conn:
|
||||
obs_ids = await _get_observation_ids(conn, bank_id)
|
||||
assert str(obs_id) not in obs_ids, "Observation should have been deleted"
|
||||
|
||||
# m3 (remaining source) should be reset for re-consolidation
|
||||
consolidated_at = await _get_consolidated_at(conn, m3)
|
||||
assert consolidated_at is None, "Remaining source memory should be reset"
|
||||
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests: delete_bank with fact_type filter
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestDeleteBankByTypeObservationCleanup:
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_clearing_experience_memories_removes_affected_observations(
|
||||
self, memory: MemoryEngine, request_context: RequestContext
|
||||
):
|
||||
"""Clearing all experience memories removes observations sourced from them."""
|
||||
bank_id = f"test-invalidate-banktype-{uuid.uuid4().hex[:8]}"
|
||||
await _ensure_bank(memory, bank_id, request_context)
|
||||
|
||||
pool = await memory._get_pool()
|
||||
async with pool.acquire() as conn:
|
||||
exp1 = await _insert_memory(conn, bank_id, "Alice went hiking last week.", "experience")
|
||||
world1 = await _insert_memory(conn, bank_id, "Alice is a hiker.", "world")
|
||||
obs_id = await _insert_observation(
|
||||
conn, bank_id, "Alice is a regular hiker.", [exp1, world1]
|
||||
)
|
||||
|
||||
# Patch out consolidation so it doesn't re-set consolidated_at before we can check it
|
||||
with patch.object(memory, "submit_async_consolidation", new=AsyncMock()):
|
||||
await memory.delete_bank(bank_id, fact_type="experience", request_context=request_context)
|
||||
|
||||
async with pool.acquire() as conn:
|
||||
obs_ids = await _get_observation_ids(conn, bank_id)
|
||||
assert str(obs_id) not in obs_ids, "Observation should have been deleted"
|
||||
|
||||
# world1 (remaining source) should be reset for re-consolidation
|
||||
consolidated_at = await _get_consolidated_at(conn, world1)
|
||||
assert consolidated_at is None, "World memory should be reset for re-consolidation"
|
||||
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_clearing_unrelated_type_leaves_observations_intact(
|
||||
self, memory: MemoryEngine, request_context: RequestContext
|
||||
):
|
||||
"""Clearing memories of a type that is not a source of any observation leaves observations untouched."""
|
||||
bank_id = f"test-invalidate-banktype-noop-{uuid.uuid4().hex[:8]}"
|
||||
await _ensure_bank(memory, bank_id, request_context)
|
||||
|
||||
pool = await memory._get_pool()
|
||||
async with pool.acquire() as conn:
|
||||
world1 = await _insert_memory(conn, bank_id, "Alice is a hiker.", "world")
|
||||
obs_id = await _insert_observation(conn, bank_id, "Alice is a regular hiker.", [world1])
|
||||
|
||||
# Deleting 'experience' type should not affect observations sourced only from 'world'
|
||||
await memory.delete_bank(bank_id, fact_type="experience", request_context=request_context)
|
||||
|
||||
async with pool.acquire() as conn:
|
||||
obs_ids = await _get_observation_ids(conn, bank_id)
|
||||
assert str(obs_id) in obs_ids, "Observation should remain untouched"
|
||||
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tests: clear_observations_for_memory
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class TestClearObservationsForMemory:
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_clears_observations_and_resets_all_source_memories(
|
||||
self, memory: MemoryEngine, request_context: RequestContext
|
||||
):
|
||||
"""Clearing observations for a memory deletes them and resets all related source memories."""
|
||||
bank_id = f"test-clear-obs-mem-{uuid.uuid4().hex[:8]}"
|
||||
await _ensure_bank(memory, bank_id, request_context)
|
||||
|
||||
pool = await memory._get_pool()
|
||||
async with pool.acquire() as conn:
|
||||
m1 = await _insert_memory(conn, bank_id, "Alice loves hiking.")
|
||||
m2 = await _insert_memory(conn, bank_id, "Alice hikes every weekend.")
|
||||
obs_id = await _insert_observation(conn, bank_id, "Alice is an avid hiker.", [m1, m2])
|
||||
|
||||
# Patch out consolidation so it doesn't re-set consolidated_at before we can check it
|
||||
with patch.object(memory, "submit_async_consolidation", new=AsyncMock()):
|
||||
result = await memory.clear_observations_for_memory(
|
||||
bank_id, str(m1), request_context=request_context
|
||||
)
|
||||
|
||||
assert result["deleted_count"] == 1
|
||||
|
||||
async with pool.acquire() as conn:
|
||||
obs_ids = await _get_observation_ids(conn, bank_id)
|
||||
assert str(obs_id) not in obs_ids, "Observation should be deleted"
|
||||
|
||||
# Both m1 (target) and m2 (remaining source) should be reset
|
||||
assert await _get_consolidated_at(conn, m1) is None, "Target memory should be reset"
|
||||
assert await _get_consolidated_at(conn, m2) is None, "Remaining source should be reset"
|
||||
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_no_observations_returns_zero(
|
||||
self, memory: MemoryEngine, request_context: RequestContext
|
||||
):
|
||||
"""Returns 0 when the memory has no associated observations."""
|
||||
bank_id = f"test-clear-obs-noop-{uuid.uuid4().hex[:8]}"
|
||||
await _ensure_bank(memory, bank_id, request_context)
|
||||
|
||||
pool = await memory._get_pool()
|
||||
async with pool.acquire() as conn:
|
||||
m1 = await _insert_memory(conn, bank_id, "Alice loves hiking.")
|
||||
|
||||
result = await memory.clear_observations_for_memory(
|
||||
bank_id, str(m1), request_context=request_context
|
||||
)
|
||||
|
||||
assert result["deleted_count"] == 0
|
||||
|
||||
async with pool.acquire() as conn:
|
||||
# Memory should still be consolidated (no observations were cleared)
|
||||
assert await _get_consolidated_at(conn, m1) is not None
|
||||
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_only_clears_observations_referencing_target_memory(
|
||||
self, memory: MemoryEngine, request_context: RequestContext
|
||||
):
|
||||
"""Clearing observations for m1 does not affect observations that only reference m2."""
|
||||
bank_id = f"test-clear-obs-selective-{uuid.uuid4().hex[:8]}"
|
||||
await _ensure_bank(memory, bank_id, request_context)
|
||||
|
||||
pool = await memory._get_pool()
|
||||
async with pool.acquire() as conn:
|
||||
m1 = await _insert_memory(conn, bank_id, "Alice loves hiking.")
|
||||
m2 = await _insert_memory(conn, bank_id, "Alice hikes every weekend.")
|
||||
m3 = await _insert_memory(conn, bank_id, "Alice climbed a mountain.")
|
||||
|
||||
obs1_id = await _insert_observation(conn, bank_id, "Alice is an avid hiker.", [m1, m2])
|
||||
obs2_id = await _insert_observation(conn, bank_id, "Alice is a mountaineer.", [m3])
|
||||
|
||||
result = await memory.clear_observations_for_memory(
|
||||
bank_id, str(m1), request_context=request_context
|
||||
)
|
||||
|
||||
assert result["deleted_count"] == 1
|
||||
|
||||
async with pool.acquire() as conn:
|
||||
obs_ids = await _get_observation_ids(conn, bank_id)
|
||||
assert str(obs1_id) not in obs_ids, "obs1 (references m1) should be deleted"
|
||||
assert str(obs2_id) in obs_ids, "obs2 (does not reference m1) should remain"
|
||||
|
||||
# m3 should still be consolidated
|
||||
assert await _get_consolidated_at(conn, m3) is not None
|
||||
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_multiple_observations_for_same_memory_all_cleared(
|
||||
self, memory: MemoryEngine, request_context: RequestContext
|
||||
):
|
||||
"""All observations referencing the target memory are cleared in one call."""
|
||||
bank_id = f"test-clear-obs-multi-{uuid.uuid4().hex[:8]}"
|
||||
await _ensure_bank(memory, bank_id, request_context)
|
||||
|
||||
pool = await memory._get_pool()
|
||||
async with pool.acquire() as conn:
|
||||
m1 = await _insert_memory(conn, bank_id, "Alice loves hiking.")
|
||||
m2 = await _insert_memory(conn, bank_id, "Alice hikes every weekend.")
|
||||
|
||||
obs1_id = await _insert_observation(conn, bank_id, "Alice hikes often.", [m1])
|
||||
obs2_id = await _insert_observation(conn, bank_id, "Alice is outdoorsy.", [m1, m2])
|
||||
|
||||
# Patch out consolidation so it doesn't re-set consolidated_at before we can check it
|
||||
with patch.object(memory, "submit_async_consolidation", new=AsyncMock()):
|
||||
result = await memory.clear_observations_for_memory(
|
||||
bank_id, str(m1), request_context=request_context
|
||||
)
|
||||
|
||||
assert result["deleted_count"] == 2
|
||||
|
||||
async with pool.acquire() as conn:
|
||||
obs_ids = await _get_observation_ids(conn, bank_id)
|
||||
assert str(obs1_id) not in obs_ids
|
||||
assert str(obs2_id) not in obs_ids
|
||||
|
||||
# m1 and m2 should both be reset
|
||||
assert await _get_consolidated_at(conn, m1) is None
|
||||
assert await _get_consolidated_at(conn, m2) is None
|
||||
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
@@ -93,7 +93,7 @@ def test_per_operation_provider_default_model():
|
||||
config = HindsightConfig.from_env()
|
||||
|
||||
# Global LLM should use OpenAI default
|
||||
assert config.llm_model == "o3-mini", f"Expected o3-mini, got {config.llm_model}"
|
||||
assert config.llm_model == "gpt-4o-mini", f"Expected gpt-4o-mini, got {config.llm_model}"
|
||||
|
||||
# Retain should use Anthropic default
|
||||
assert (
|
||||
|
||||
@@ -7,14 +7,17 @@ These tests verify:
|
||||
3. Recovery from tool execution errors
|
||||
"""
|
||||
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
from hindsight_api.engine.reflect.agent import (
|
||||
_normalize_tool_name,
|
||||
_is_done_tool,
|
||||
_clean_answer_text,
|
||||
_clean_done_answer,
|
||||
_count_messages_tokens,
|
||||
_is_context_overflow_error,
|
||||
_is_done_tool,
|
||||
_normalize_tool_name,
|
||||
run_reflect_agent,
|
||||
)
|
||||
from hindsight_api.engine.response_models import LLMToolCall, LLMToolCallResult, TokenUsage
|
||||
@@ -412,3 +415,193 @@ class TestReflectAgentMocked:
|
||||
# Should have a result even if no memories found
|
||||
assert result is not None
|
||||
assert result.iterations == 3
|
||||
|
||||
|
||||
class TestContextOverflowHelpers:
|
||||
"""Unit tests for context-overflow detection helpers."""
|
||||
|
||||
def test_count_messages_tokens_basic(self):
|
||||
messages = [
|
||||
{"role": "system", "content": "You are a helpful assistant."},
|
||||
{"role": "user", "content": "What is the capital of France?"},
|
||||
]
|
||||
count = _count_messages_tokens(messages)
|
||||
assert count > 0
|
||||
# Rough sanity check: ~10 tokens for each message
|
||||
assert count < 100
|
||||
|
||||
def test_count_messages_tokens_with_tool_result(self):
|
||||
"""A large tool result should substantially increase the count."""
|
||||
small_messages = [{"role": "user", "content": "hi"}]
|
||||
large_messages = [
|
||||
{"role": "user", "content": "hi"},
|
||||
{
|
||||
"role": "tool",
|
||||
"tool_call_id": "x",
|
||||
"name": "recall",
|
||||
"content": '{"memories": [' + ', '.join([f'{{"id": "m{i}", "content": "A long memory fact about some topic that goes on and on."}}' for i in range(50)]) + ']}',
|
||||
},
|
||||
]
|
||||
small = _count_messages_tokens(small_messages)
|
||||
large = _count_messages_tokens(large_messages)
|
||||
assert large > small + 200
|
||||
|
||||
def test_is_context_overflow_error_openai(self):
|
||||
assert _is_context_overflow_error(Exception("context_length_exceeded: too many tokens"))
|
||||
assert _is_context_overflow_error(Exception("This model's maximum context length is 128000 tokens. However, your messages resulted in 142164 tokens."))
|
||||
|
||||
def test_is_context_overflow_error_anthropic(self):
|
||||
assert _is_context_overflow_error(Exception("prompt_too_long"))
|
||||
assert _is_context_overflow_error(Exception("prompt is too long for this model"))
|
||||
|
||||
def test_is_context_overflow_error_gemini(self):
|
||||
assert _is_context_overflow_error(Exception("RESOURCE_EXHAUSTED: quota exceeded"))
|
||||
|
||||
def test_is_context_overflow_error_generic(self):
|
||||
assert _is_context_overflow_error(Exception("input is too long to process"))
|
||||
assert _is_context_overflow_error(Exception("too many tokens in the request"))
|
||||
|
||||
def test_is_context_overflow_error_unrelated(self):
|
||||
assert not _is_context_overflow_error(Exception("connection timeout"))
|
||||
assert not _is_context_overflow_error(Exception("rate limit exceeded"))
|
||||
assert not _is_context_overflow_error(ValueError("invalid argument"))
|
||||
|
||||
|
||||
class TestContextOverflowBehavior:
|
||||
"""Test that the reflect agent handles context overflow gracefully."""
|
||||
|
||||
@pytest.fixture
|
||||
def mock_llm(self):
|
||||
llm = MagicMock()
|
||||
llm.call_with_tools = AsyncMock()
|
||||
llm.call = AsyncMock(
|
||||
return_value=("Synthesized answer from gathered evidence.", TokenUsage(input_tokens=50, output_tokens=20, total_tokens=70))
|
||||
)
|
||||
return llm
|
||||
|
||||
@pytest.fixture
|
||||
def mock_functions_with_large_output(self):
|
||||
"""Mock functions that return a large enough payload to exceed a tiny token budget."""
|
||||
large_memories = [
|
||||
{"id": f"mem-{i}", "content": f"Memory fact number {i}: " + "A" * 200}
|
||||
for i in range(20)
|
||||
]
|
||||
return {
|
||||
"search_mental_models_fn": AsyncMock(return_value={"mental_models": []}),
|
||||
"search_observations_fn": AsyncMock(return_value={"observations": []}),
|
||||
"recall_fn": AsyncMock(return_value={"memories": large_memories}),
|
||||
"expand_fn": AsyncMock(return_value={"memories": []}),
|
||||
}
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_proactive_guard_fires_when_budget_exceeded(self, mock_llm, mock_functions_with_large_output):
|
||||
"""When token count exceeds max_context_tokens after a tool call, the agent
|
||||
should immediately synthesize from gathered evidence instead of making
|
||||
another LLM call that would overflow."""
|
||||
# First call: LLM calls recall (forced by iter 0 with no mental models)
|
||||
mock_llm.call_with_tools.return_value = LLMToolCallResult(
|
||||
tool_calls=[LLMToolCall(id="1", name="recall", arguments={"query": "test"})],
|
||||
finish_reason="tool_calls",
|
||||
)
|
||||
|
||||
# Set a tiny token budget — the recall result alone will blow past it
|
||||
result = await run_reflect_agent(
|
||||
llm_config=mock_llm,
|
||||
bank_id="test-bank",
|
||||
query="What do you know?",
|
||||
bank_profile={"name": "Test", "mission": "Testing"},
|
||||
max_context_tokens=100,
|
||||
**mock_functions_with_large_output,
|
||||
)
|
||||
|
||||
assert result.text == "Synthesized answer from gathered evidence."
|
||||
# call_with_tools was called once (for the forced recall), then the guard
|
||||
# kicked in — no further tool-call iterations
|
||||
assert mock_llm.call_with_tools.call_count == 1
|
||||
# llm.call() was invoked to generate the final synthesis
|
||||
mock_llm.call.assert_called_once()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_context_overflow_error_skips_retry(self, mock_llm, mock_functions_with_large_output):
|
||||
"""A context_length_exceeded error from the LLM should NOT be retried —
|
||||
it should immediately fall back to final synthesis."""
|
||||
mock_llm.call_with_tools.side_effect = Exception(
|
||||
"context_length_exceeded: messages resulted in 150000 tokens."
|
||||
)
|
||||
|
||||
result = await run_reflect_agent(
|
||||
llm_config=mock_llm,
|
||||
bank_id="test-bank",
|
||||
query="What do you know?",
|
||||
bank_profile={"name": "Test", "mission": "Testing"},
|
||||
max_iterations=5,
|
||||
**mock_functions_with_large_output,
|
||||
)
|
||||
|
||||
assert result is not None
|
||||
# Should have attempted only 1 iteration (no retry on overflow error)
|
||||
assert mock_llm.call_with_tools.call_count == 1
|
||||
# Final synthesis was called
|
||||
mock_llm.call.assert_called_once()
|
||||
|
||||
|
||||
class TestContextOverflowIntegration:
|
||||
"""Integration test: real LLM with a very small max_context_tokens.
|
||||
|
||||
The agent will make one real LLM call (forced tool choice), receive a large
|
||||
tool result that exceeds the tiny budget, then synthesize from it via a second
|
||||
real LLM call — all without raising a context_length_exceeded error.
|
||||
"""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_reflect_completes_with_tiny_context_budget(self, memory, request_context):
|
||||
"""End-to-end: reflect on a bank with max_context_tokens=1 (tiny budget).
|
||||
|
||||
Setting max_context_tokens=1 guarantees the proactive guard fires as soon
|
||||
as the first tool result is received and evidence is available.
|
||||
The result must be a non-empty string with no exception raised.
|
||||
"""
|
||||
import uuid
|
||||
from unittest.mock import patch
|
||||
|
||||
bank_id = f"test-ctx-overflow-{uuid.uuid4().hex[:8]}"
|
||||
try:
|
||||
# Retain a handful of facts so the recall tool has something to return
|
||||
await memory.retain_async(
|
||||
bank_id=bank_id,
|
||||
content="Alice is a software engineer who enjoys hiking on weekends.",
|
||||
request_context=request_context,
|
||||
)
|
||||
await memory.retain_async(
|
||||
bank_id=bank_id,
|
||||
content="Bob is a designer who loves cooking Italian food.",
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
# Patch get_config where memory_engine uses it, injecting a tiny
|
||||
# max_context_tokens. Everything else delegates to the real config.
|
||||
real_config = memory._get_raw_config() if hasattr(memory, "_get_raw_config") else None
|
||||
from hindsight_api.config import get_config as _real_get_config
|
||||
|
||||
class _TinyContextProxy:
|
||||
"""Forwards all attribute access to the real config proxy except
|
||||
reflect_max_context_tokens which is forced to 1."""
|
||||
_real = _real_get_config()
|
||||
|
||||
def __getattr__(self, name: str):
|
||||
if name == "reflect_max_context_tokens":
|
||||
return 1
|
||||
return getattr(self._real, name)
|
||||
|
||||
with patch("hindsight_api.engine.memory_engine.get_config", return_value=_TinyContextProxy()):
|
||||
result = await memory.reflect_async(
|
||||
bank_id=bank_id,
|
||||
query="Tell me about the people you know.",
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
assert result.text, "reflect must return a non-empty answer"
|
||||
assert result.usage.total_tokens > 0
|
||||
|
||||
finally:
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
@@ -0,0 +1,71 @@
|
||||
"""
|
||||
Regression test for UnboundLocalError in recall when the reranker raises.
|
||||
|
||||
Before the fix, `scored_results` and `pre_filtered_count` were only assigned
|
||||
inside the `try` block, but referenced in the `finally` block. If
|
||||
`reranker_instance.rerank()` (or `ensure_initialized()`) raised, the `finally`
|
||||
block crashed with `UnboundLocalError` instead of propagating the original
|
||||
exception.
|
||||
|
||||
Fix: initialise both variables to safe defaults before the try/finally block.
|
||||
"""
|
||||
|
||||
from datetime import datetime, timezone
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_recall_reranker_error_does_not_raise_unbound_local(memory, request_context):
|
||||
"""Recall must propagate the reranker's exception, not an UnboundLocalError."""
|
||||
bank_id = f"test_reranker_err_{datetime.now(timezone.utc).timestamp()}"
|
||||
|
||||
try:
|
||||
await memory.retain_async(
|
||||
bank_id=bank_id,
|
||||
content="Paris is the capital of France",
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
# Simulate a reranker failure (e.g. Cohere API error on empty/small candidate set)
|
||||
rerank_mock = AsyncMock(side_effect=RuntimeError("reranker API error"))
|
||||
memory._cross_encoder_reranker._initialized = True # skip ensure_initialized
|
||||
|
||||
with patch.object(memory._cross_encoder_reranker, "rerank", rerank_mock):
|
||||
with pytest.raises(Exception, match="reranker API error"):
|
||||
await memory.recall_async(
|
||||
bank_id=bank_id,
|
||||
query="capital of France",
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
finally:
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_recall_reranker_init_error_does_not_raise_unbound_local(memory, request_context):
|
||||
"""Same regression when ensure_initialized() raises (before pre_filtered_count is set)."""
|
||||
bank_id = f"test_reranker_init_err_{datetime.now(timezone.utc).timestamp()}"
|
||||
|
||||
try:
|
||||
await memory.retain_async(
|
||||
bank_id=bank_id,
|
||||
content="Paris is the capital of France",
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
init_mock = AsyncMock(side_effect=RuntimeError("reranker init failed"))
|
||||
memory._cross_encoder_reranker._initialized = False
|
||||
|
||||
with patch.object(memory._cross_encoder_reranker, "ensure_initialized", init_mock):
|
||||
with pytest.raises(Exception, match="reranker init failed"):
|
||||
await memory.recall_async(
|
||||
bank_id=bank_id,
|
||||
query="capital of France",
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
finally:
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
@@ -591,6 +591,125 @@ async def test_mentioned_at_from_context_string(memory, request_context):
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
|
||||
# ============================================================
|
||||
# No Timestamp Tests
|
||||
# ============================================================
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_retain_no_timestamp(memory, request_context):
|
||||
"""
|
||||
Test retaining content with explicit "no timestamp" sentinel.
|
||||
|
||||
When event_date=None is passed explicitly in the dict (i.e. caller opted into
|
||||
no timestamp), mentioned_at should be NULL in the DB rather than defaulting to now().
|
||||
"""
|
||||
bank_id = f"test_no_timestamp_{datetime.now(timezone.utc).timestamp()}"
|
||||
|
||||
try:
|
||||
# Use retain_batch_async with explicit event_date=None key to signal "no timestamp"
|
||||
unit_ids_list = await memory.retain_batch_async(
|
||||
bank_id=bank_id,
|
||||
contents=[
|
||||
{
|
||||
"content": "The capital of France is Paris. The Eiffel Tower is located in Paris.",
|
||||
"context": "general knowledge",
|
||||
"event_date": None, # Explicit sentinel: no timestamp
|
||||
}
|
||||
],
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
assert len(unit_ids_list) > 0, "Should create at least one batch result"
|
||||
unit_ids = unit_ids_list[0]
|
||||
assert len(unit_ids) > 0, "Should have extracted and stored facts"
|
||||
|
||||
# Recall the facts
|
||||
result = await memory.recall_async(
|
||||
bank_id=bank_id,
|
||||
query="Where is the Eiffel Tower?",
|
||||
budget=Budget.LOW,
|
||||
max_tokens=500,
|
||||
fact_type=["world"],
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
assert len(result.results) > 0, "Should recall the stored fact"
|
||||
|
||||
# All temporal fields should be None for temporally agnostic content
|
||||
for fact in result.results:
|
||||
assert fact.mentioned_at is None, (
|
||||
f"mentioned_at should be None for no-timestamp content, got {fact.mentioned_at}"
|
||||
)
|
||||
|
||||
print(f"\n✓ Test passed: mentioned_at is None for {len(result.results)} fact(s)")
|
||||
|
||||
finally:
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_retain_omit_timestamp_defaults_to_now(memory, request_context):
|
||||
"""
|
||||
Backward-compatibility regression test: omitting event_date still stores a real datetime.
|
||||
|
||||
When event_date is absent from the content dict (key not present), the orchestrator
|
||||
should default to utcnow() — preserving existing behavior.
|
||||
"""
|
||||
bank_id = f"test_default_timestamp_{datetime.now(timezone.utc).timestamp()}"
|
||||
before = datetime.now(timezone.utc)
|
||||
|
||||
try:
|
||||
# Omit event_date entirely — should default to now()
|
||||
unit_ids_list = await memory.retain_batch_async(
|
||||
bank_id=bank_id,
|
||||
contents=[
|
||||
{
|
||||
"content": "Alice is a software engineer who loves Python.",
|
||||
"context": "profile",
|
||||
# event_date intentionally omitted
|
||||
}
|
||||
],
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
after = datetime.now(timezone.utc)
|
||||
|
||||
assert len(unit_ids_list) > 0
|
||||
unit_ids = unit_ids_list[0]
|
||||
assert len(unit_ids) > 0, "Should have extracted and stored facts"
|
||||
|
||||
# Recall and verify mentioned_at is a real datetime close to now
|
||||
result = await memory.recall_async(
|
||||
bank_id=bank_id,
|
||||
query="Who is Alice?",
|
||||
budget=Budget.LOW,
|
||||
max_tokens=500,
|
||||
fact_type=["world"],
|
||||
request_context=request_context,
|
||||
)
|
||||
|
||||
assert len(result.results) > 0, "Should recall the fact"
|
||||
fact = result.results[0]
|
||||
|
||||
assert fact.mentioned_at is not None, "mentioned_at should be set when event_date is omitted"
|
||||
|
||||
if isinstance(fact.mentioned_at, str):
|
||||
mentioned_dt = datetime.fromisoformat(fact.mentioned_at.replace("Z", "+00:00"))
|
||||
else:
|
||||
mentioned_dt = fact.mentioned_at
|
||||
|
||||
# Should be within 60s of when we ran the test
|
||||
assert before <= mentioned_dt <= after + timedelta(seconds=60), (
|
||||
f"mentioned_at {mentioned_dt} should be close to now ({before} – {after})"
|
||||
)
|
||||
|
||||
print(f"\n✓ Test passed: mentioned_at={mentioned_dt} is a real datetime (backward compat)")
|
||||
|
||||
finally:
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Context Tracking Tests
|
||||
# ============================================================
|
||||
@@ -2256,3 +2375,64 @@ async def test_retain_batch_with_per_item_tags_on_document(memory, request_conte
|
||||
finally:
|
||||
await memory.delete_bank(bank_id, request_context=request_context)
|
||||
print(f"\n=== Cleaned up bank: {bank_id} ===")
|
||||
|
||||
|
||||
def test_retain_mission_injected_into_prompt():
|
||||
"""Test that retain_mission is injected as a FOCUS section into any extraction mode."""
|
||||
from unittest.mock import MagicMock
|
||||
from hindsight_api.engine.retain.fact_extraction import _build_extraction_prompt_and_schema
|
||||
|
||||
spec = "Focus on technical decisions and architecture choices only."
|
||||
|
||||
# Test with concise mode
|
||||
config = MagicMock()
|
||||
config.retain_extraction_mode = "concise"
|
||||
config.retain_mission = spec
|
||||
config.retain_custom_instructions = None
|
||||
config.retain_extract_causal_links = False
|
||||
|
||||
prompt, _ = _build_extraction_prompt_and_schema(config)
|
||||
assert spec in prompt
|
||||
assert "FOCUS" in prompt
|
||||
|
||||
# retain_mission is present regardless of extraction mode (verbose has its own template, no spec injection)
|
||||
config.retain_extraction_mode = "verbose"
|
||||
prompt_verbose, _ = _build_extraction_prompt_and_schema(config)
|
||||
# verbose uses its own template - spec not injected there
|
||||
assert spec not in prompt_verbose
|
||||
|
||||
|
||||
def test_retain_mission_absent_when_not_set():
|
||||
"""Test that no FOCUS section appears when retain_mission is not set."""
|
||||
from unittest.mock import MagicMock
|
||||
from hindsight_api.engine.retain.fact_extraction import _build_extraction_prompt_and_schema
|
||||
|
||||
config = MagicMock()
|
||||
config.retain_extraction_mode = "concise"
|
||||
config.retain_mission = None
|
||||
config.retain_custom_instructions = None
|
||||
config.retain_extract_causal_links = False
|
||||
|
||||
prompt, _ = _build_extraction_prompt_and_schema(config)
|
||||
assert "FOCUS" not in prompt
|
||||
assert "retain_mission_section" not in prompt
|
||||
|
||||
|
||||
def test_retain_mission_config_loaded_from_env():
|
||||
"""Test that retain_mission is loaded from env and is a configurable field."""
|
||||
import os
|
||||
from hindsight_api.config import HindsightConfig, _get_raw_config, clear_config_cache
|
||||
|
||||
original = os.getenv("HINDSIGHT_API_RETAIN_MISSION")
|
||||
try:
|
||||
os.environ["HINDSIGHT_API_RETAIN_MISSION"] = "Only technical decisions."
|
||||
clear_config_cache()
|
||||
config = _get_raw_config()
|
||||
assert config.retain_mission == "Only technical decisions."
|
||||
assert "retain_mission" in HindsightConfig.get_configurable_fields()
|
||||
finally:
|
||||
if original is None:
|
||||
os.environ.pop("HINDSIGHT_API_RETAIN_MISSION", None)
|
||||
else:
|
||||
os.environ["HINDSIGHT_API_RETAIN_MISSION"] = original
|
||||
clear_config_cache()
|
||||
|
||||
@@ -233,6 +233,35 @@ class TestFilterResultsByTags:
|
||||
assert len(filtered) == 1
|
||||
assert filtered[0].tags == ["a", "b", "c"] # Has a, b, AND c
|
||||
|
||||
def test_all_strict_superset_observation_matches_incoming_memory_tags(self):
|
||||
"""
|
||||
Consolidation scenario: an incoming memory with tags ['user:bob', 'session:id1']
|
||||
uses all_strict matching to find existing observations.
|
||||
|
||||
An observation tagged ['user:bob', 'session:id1', 'place:online'] IS matched
|
||||
because it contains all of the incoming memory's tags (superset).
|
||||
This is NOT exact matching — an observation with extra tags is still a valid match.
|
||||
"""
|
||||
# Incoming memory tags (e.g. from a new retain call)
|
||||
incoming_tags = ["user:bob", "session:id1"]
|
||||
|
||||
# Candidate observations with different tag sets
|
||||
exact_match = MockResult(["user:bob", "session:id1"])
|
||||
superset_match = MockResult(["session:id1", "user:bob", "place:online"])
|
||||
different_user = MockResult(["user:alice", "session:id1"])
|
||||
missing_session = MockResult(["user:bob"])
|
||||
|
||||
results = [exact_match, superset_match, different_user, missing_session]
|
||||
filtered = filter_results_by_tags(results, incoming_tags, match="all_strict")
|
||||
|
||||
# Both exact_match and superset_match have all incoming tags → both match
|
||||
assert len(filtered) == 2
|
||||
assert exact_match in filtered
|
||||
assert superset_match in filtered
|
||||
# different_user and missing_session are excluded because they lack at least one tag
|
||||
assert different_user not in filtered
|
||||
assert missing_session not in filtered
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# Integration Tests for tags in retain/recall/reflect
|
||||
@@ -890,3 +919,40 @@ async def test_list_tags_ordered_by_count(api_client):
|
||||
# common (3) should come before medium (2) which should come before rare (1)
|
||||
assert tags.index("common") < tags.index("medium")
|
||||
assert tags.index("medium") < tags.index("rare")
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_list_memories_includes_tags(api_client, test_bank_id):
|
||||
"""Test that list memories endpoint returns tags for each memory unit.
|
||||
|
||||
Regression test: tags were previously omitted from the SELECT query in
|
||||
list_memory_units, causing the memory dialog in the UI to show no tags
|
||||
even when memories had been stored with tags.
|
||||
"""
|
||||
tags = ["user_alice", "session_xyz", "project_alpha", "team_eng", "env_prod", "region_us"]
|
||||
|
||||
response = await api_client.post(
|
||||
f"/v1/default/banks/{test_bank_id}/memories",
|
||||
json={
|
||||
"items": [
|
||||
{
|
||||
"content": "Alice is a senior engineer on the platform team.",
|
||||
"tags": tags,
|
||||
}
|
||||
]
|
||||
},
|
||||
)
|
||||
assert response.status_code == 200
|
||||
|
||||
# List memories and verify all tags are returned
|
||||
response = await api_client.get(f"/v1/default/banks/{test_bank_id}/memories/list")
|
||||
assert response.status_code == 200
|
||||
result = response.json()
|
||||
|
||||
assert result["total"] > 0
|
||||
memory_item = next((item for item in result["items"] if "Alice" in item["text"]), None)
|
||||
assert memory_item is not None, "Should find the stored memory"
|
||||
assert "tags" in memory_item, "Memory item must include a 'tags' field"
|
||||
assert set(memory_item["tags"]) == set(tags), (
|
||||
f"All {len(tags)} tags should be returned, got: {memory_item['tags']}"
|
||||
)
|
||||
|
||||
@@ -266,9 +266,10 @@ def test_llm_span_recorder_provider_mapping(mock_time):
|
||||
# ==================== Parent Span Tests ====================
|
||||
|
||||
|
||||
@patch("hindsight_api.tracing._tracing_enabled", False)
|
||||
def test_create_operation_span_disabled():
|
||||
"""Test that create_operation_span returns no-op when tracing is disabled."""
|
||||
# Tracing should be disabled by default
|
||||
# Tracing should be disabled by default (explicitly patched for test isolation)
|
||||
assert not is_tracing_enabled()
|
||||
|
||||
# Should return a no-op context manager
|
||||
|
||||
@@ -73,7 +73,10 @@ def test_llm_wrapper_vertexai_adc_auth():
|
||||
|
||||
with patch.dict(
|
||||
os.environ,
|
||||
{"HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID": "test-project"},
|
||||
{
|
||||
"HINDSIGHT_API_LLM_VERTEXAI_PROJECT_ID": "test-project",
|
||||
"HINDSIGHT_API_LLM_VERTEXAI_SERVICE_ACCOUNT_KEY": "", # Clear SA key to test ADC path
|
||||
},
|
||||
clear=False,
|
||||
):
|
||||
from hindsight_api.config import clear_config_cache
|
||||
@@ -96,11 +99,10 @@ def test_llm_wrapper_vertexai_adc_auth():
|
||||
assert provider._gemini_client is not None
|
||||
|
||||
# Verify genai.Client was called with vertexai=True
|
||||
mock_client_cls.assert_called_once_with(
|
||||
vertexai=True,
|
||||
project="test-project",
|
||||
location="us-central1",
|
||||
)
|
||||
call_kwargs = mock_client_cls.call_args.kwargs
|
||||
assert call_kwargs["vertexai"] is True
|
||||
assert call_kwargs["project"] == "test-project"
|
||||
assert call_kwargs["location"] == "us-central1"
|
||||
|
||||
clear_config_cache()
|
||||
|
||||
@@ -141,12 +143,11 @@ def test_llm_wrapper_vertexai_sa_auth():
|
||||
assert provider._gemini_client is not None
|
||||
|
||||
# Verify credentials were passed to genai.Client
|
||||
mock_client_cls.assert_called_once_with(
|
||||
vertexai=True,
|
||||
project="test-project",
|
||||
location="us-central1",
|
||||
credentials=mock_credentials,
|
||||
)
|
||||
call_kwargs = mock_client_cls.call_args.kwargs
|
||||
assert call_kwargs["vertexai"] is True
|
||||
assert call_kwargs["project"] == "test-project"
|
||||
assert call_kwargs["location"] == "us-central1"
|
||||
assert call_kwargs["credentials"] is mock_credentials
|
||||
|
||||
clear_config_cache()
|
||||
|
||||
|
||||
@@ -268,24 +268,26 @@ class TestWorkerPoller:
|
||||
assert row["completed_at"] is not None
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_executor_exception_does_not_crash_poller(self, pool, clean_operations):
|
||||
"""Test that unexpected exceptions from executor are caught and don't crash the poller.
|
||||
async def test_executor_exception_triggers_retry(self, pool, clean_operations):
|
||||
"""Test that exceptions from the executor trigger _retry_or_fail (not a crash).
|
||||
|
||||
If the executor raises an unexpected exception (which MemoryEngine.execute_task should NOT do,
|
||||
but could happen from schema setup or other infrastructure issues), the poller should catch it
|
||||
gracefully. Status remains 'processing' since neither executor nor poller handled it.
|
||||
When the executor re-raises an exception (as MemoryEngine.execute_task does for
|
||||
retryable task failures), the poller calls _retry_or_fail, which resets the task
|
||||
back to 'pending' and increments retry_count so it can be reclaimed.
|
||||
|
||||
This is the fix for the consolidation deadlock: previously submit_task was called
|
||||
with only a task_payload update, leaving status='processing' forever.
|
||||
"""
|
||||
from hindsight_api.worker import WorkerPoller
|
||||
from hindsight_api.worker.poller import ClaimedTask
|
||||
|
||||
# Create a pending task
|
||||
bank_id = f"test-worker-{uuid.uuid4().hex[:8]}"
|
||||
op_id = uuid.uuid4()
|
||||
payload = json.dumps({"type": "test_task", "operation_id": str(op_id), "bank_id": bank_id})
|
||||
payload = json.dumps({"type": "consolidation", "operation_id": str(op_id), "bank_id": bank_id})
|
||||
await pool.execute(
|
||||
"""
|
||||
INSERT INTO async_operations (operation_id, bank_id, operation_type, status, task_payload, worker_id)
|
||||
VALUES ($1, $2, 'test', 'processing', $3::jsonb, 'test-worker-1')
|
||||
INSERT INTO async_operations (operation_id, bank_id, operation_type, status, task_payload, worker_id, claimed_at)
|
||||
VALUES ($1, $2, 'consolidation', 'processing', $3::jsonb, 'test-worker-1', now())
|
||||
""",
|
||||
op_id,
|
||||
bank_id,
|
||||
@@ -293,43 +295,100 @@ class TestWorkerPoller:
|
||||
)
|
||||
|
||||
async def failing_executor(task_dict):
|
||||
raise ValueError("Unexpected infrastructure failure")
|
||||
raise ValueError("TimeoutError during recall")
|
||||
|
||||
poller = WorkerPoller(
|
||||
pool=pool,
|
||||
worker_id="test-worker-1",
|
||||
executor=failing_executor,
|
||||
max_retries=3,
|
||||
)
|
||||
|
||||
# Execute - should catch exception without crashing
|
||||
task_dict = json.loads(payload)
|
||||
claimed_task = ClaimedTask(operation_id=str(op_id), task_dict=task_dict, schema=None)
|
||||
await poller.execute_task(claimed_task)
|
||||
|
||||
# Wait for background task to complete
|
||||
completed = await poller.wait_for_active_tasks(timeout=5.0)
|
||||
assert completed, "Task did not complete within timeout"
|
||||
|
||||
# Status stays 'processing' since the poller no longer manages status
|
||||
# Task must be reset to 'pending' with worker_id/claimed_at cleared — not left as
|
||||
# 'processing', which would cause a permanent deadlock via the NOT EXISTS guard.
|
||||
row = await pool.fetchrow(
|
||||
"SELECT status FROM async_operations WHERE operation_id = $1",
|
||||
"SELECT status, worker_id, claimed_at, retry_count FROM async_operations WHERE operation_id = $1",
|
||||
op_id,
|
||||
)
|
||||
assert row["status"] == "processing"
|
||||
assert row["status"] == "pending", (
|
||||
f"REGRESSION: Task status is '{row['status']}' instead of 'pending'. "
|
||||
"A task stuck in 'processing' after a retry causes a consolidation deadlock."
|
||||
)
|
||||
assert row["worker_id"] is None, "worker_id must be cleared on retry"
|
||||
assert row["claimed_at"] is None, "claimed_at must be cleared on retry"
|
||||
assert row["retry_count"] == 1
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_executor_exception_marks_failed_after_max_retries(self, pool, clean_operations):
|
||||
"""Test that a task is permanently marked 'failed' once retry_count hits max_retries.
|
||||
|
||||
After max_retries exhaustion the task must NOT be reset to 'pending' — it should
|
||||
be marked 'failed' with an error message so it stops consuming retry budget.
|
||||
"""
|
||||
from hindsight_api.worker import WorkerPoller
|
||||
from hindsight_api.worker.poller import ClaimedTask
|
||||
|
||||
max_retries = 3
|
||||
bank_id = f"test-worker-{uuid.uuid4().hex[:8]}"
|
||||
op_id = uuid.uuid4()
|
||||
payload = json.dumps({"type": "consolidation", "operation_id": str(op_id), "bank_id": bank_id})
|
||||
# Insert with retry_count already at the limit
|
||||
await pool.execute(
|
||||
"""
|
||||
INSERT INTO async_operations (operation_id, bank_id, operation_type, status, task_payload, worker_id, claimed_at, retry_count)
|
||||
VALUES ($1, $2, 'consolidation', 'processing', $3::jsonb, 'test-worker-1', now(), $4)
|
||||
""",
|
||||
op_id,
|
||||
bank_id,
|
||||
payload,
|
||||
max_retries,
|
||||
)
|
||||
|
||||
async def failing_executor(task_dict):
|
||||
raise ValueError("Still failing after all retries")
|
||||
|
||||
poller = WorkerPoller(
|
||||
pool=pool,
|
||||
worker_id="test-worker-1",
|
||||
executor=failing_executor,
|
||||
max_retries=max_retries,
|
||||
)
|
||||
|
||||
task_dict = json.loads(payload)
|
||||
claimed_task = ClaimedTask(operation_id=str(op_id), task_dict=task_dict, schema=None)
|
||||
await poller.execute_task(claimed_task)
|
||||
|
||||
completed = await poller.wait_for_active_tasks(timeout=5.0)
|
||||
assert completed, "Task did not complete within timeout"
|
||||
|
||||
row = await pool.fetchrow(
|
||||
"SELECT status, error_message, retry_count FROM async_operations WHERE operation_id = $1",
|
||||
op_id,
|
||||
)
|
||||
assert row["status"] == "failed", (
|
||||
f"Expected 'failed' after max retries, got '{row['status']}'"
|
||||
)
|
||||
assert row["error_message"] is not None
|
||||
assert "Max retries" in row["error_message"]
|
||||
assert row["retry_count"] == max_retries # not incremented further
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_executor_failed_status_not_overridden(self, pool, clean_operations):
|
||||
"""REGRESSION TEST: Verify poller does NOT overwrite executor's 'failed' status to 'completed'.
|
||||
|
||||
This test catches the bug where the poller always called _mark_completed() after executor
|
||||
returned, overwriting the 'failed' status that the executor had already set.
|
||||
|
||||
Scenario:
|
||||
1. Executor catches an internal error and marks the operation as 'failed' in the DB
|
||||
2. Executor returns normally (does NOT re-raise) - this is how MemoryEngine.execute_task works
|
||||
This test covers the non-retryable failure path (e.g., file_convert_retain):
|
||||
1. Executor catches an internal error, marks the operation as 'failed' in the DB
|
||||
2. Executor returns normally (does NOT re-raise) — so no exception reaches the poller
|
||||
3. The poller must NOT overwrite the 'failed' status to 'completed'
|
||||
|
||||
With the old buggy code, this test would FAIL (status would be 'completed').
|
||||
Retryable failures re-raise instead (see test_executor_exception_triggers_retry).
|
||||
"""
|
||||
from hindsight_api.worker import WorkerPoller
|
||||
from hindsight_api.worker.poller import ClaimedTask
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "hindsight-cli"
|
||||
version = "0.4.11"
|
||||
version = "0.4.14"
|
||||
edition = "2021"
|
||||
authors = ["Hindsight Team"]
|
||||
description = "A beautiful CLI for Hindsight - semantic memory system"
|
||||
|
||||
@@ -127,6 +127,7 @@ impl ApiClient {
|
||||
mission: None,
|
||||
background: None,
|
||||
disposition: None,
|
||||
..Default::default()
|
||||
};
|
||||
let response = self.client.create_or_update_bank(agent_id, None, &request).await?;
|
||||
Ok(response.into_inner())
|
||||
@@ -436,6 +437,7 @@ impl ApiClient {
|
||||
mission: Some(mission.to_string()),
|
||||
background: None,
|
||||
disposition: None,
|
||||
..Default::default()
|
||||
};
|
||||
let response = self.client.update_bank(bank_id, None, &request).await?;
|
||||
Ok(response.into_inner())
|
||||
@@ -450,7 +452,7 @@ impl ApiClient {
|
||||
_verbose: bool,
|
||||
) -> Result<types::GraphDataResponse> {
|
||||
self.runtime.block_on(async {
|
||||
let response = self.client.get_graph(bank_id, limit, type_filter, None).await?;
|
||||
let response = self.client.get_graph(bank_id, limit, type_filter, None, None, None, None).await?;
|
||||
Ok(response.into_inner())
|
||||
})
|
||||
}
|
||||
|
||||
@@ -293,6 +293,7 @@ pub fn create(
|
||||
mission: mission_text,
|
||||
background: None,
|
||||
disposition,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let response = client.create_bank(bank_id, &request, verbose);
|
||||
@@ -356,6 +357,7 @@ pub fn update(
|
||||
mission: mission_text,
|
||||
background: None,
|
||||
disposition,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let response = client.update_bank(bank_id, &request, verbose);
|
||||
|
||||
@@ -266,6 +266,7 @@ pub fn recall(
|
||||
max_tokens: chunk_max_tokens,
|
||||
}),
|
||||
entities: None,
|
||||
source_facts: None,
|
||||
})
|
||||
} else {
|
||||
None
|
||||
@@ -386,6 +387,7 @@ pub fn retain(
|
||||
document_id: Some(doc_id.clone()),
|
||||
entities: None,
|
||||
tags: None,
|
||||
observation_scopes: None,
|
||||
};
|
||||
|
||||
let request = RetainRequest {
|
||||
|
||||
@@ -67,7 +67,7 @@ fn format_error_message(err: &anyhow::Error, api_url: &str) -> String {
|
||||
"Bank configuration API is disabled".bright_red().bold(),
|
||||
"API URL:".bright_yellow(),
|
||||
api_url.bright_white(),
|
||||
"This feature is disabled by default for security.".bright_yellow(),
|
||||
"This feature has been disabled on the server.".bright_yellow(),
|
||||
"To enable, set HINDSIGHT_API_ENABLE_BANK_CONFIG_API=true on the API server".bright_white(),
|
||||
"Note:".bright_cyan(),
|
||||
"This allows per-bank LLM configuration overrides via API".bright_white()
|
||||
|
||||
@@ -7,7 +7,7 @@ info:
|
||||
name: Apache 2.0
|
||||
url: https://www.apache.org/licenses/LICENSE-2.0.html
|
||||
title: Hindsight HTTP API
|
||||
version: 0.4.11
|
||||
version: 0.4.14
|
||||
servers:
|
||||
- url: /
|
||||
paths:
|
||||
@@ -83,6 +83,34 @@ paths:
|
||||
title: Limit
|
||||
type: integer
|
||||
style: form
|
||||
- explode: true
|
||||
in: query
|
||||
name: q
|
||||
required: false
|
||||
schema:
|
||||
nullable: true
|
||||
type: string
|
||||
style: form
|
||||
- explode: true
|
||||
in: query
|
||||
name: tags
|
||||
required: false
|
||||
schema:
|
||||
items:
|
||||
nullable: true
|
||||
type: string
|
||||
nullable: true
|
||||
type: array
|
||||
style: form
|
||||
- explode: true
|
||||
in: query
|
||||
name: tags_match
|
||||
required: false
|
||||
schema:
|
||||
default: all_strict
|
||||
title: Tags Match
|
||||
type: string
|
||||
style: form
|
||||
- explode: false
|
||||
in: header
|
||||
name: authorization
|
||||
@@ -1564,6 +1592,7 @@ paths:
|
||||
- Operations
|
||||
/v1/default/banks/{bank_id}/profile:
|
||||
get:
|
||||
deprecated: true
|
||||
description: Get disposition traits and mission for a memory bank. Auto-creates
|
||||
agent with defaults if not exists.
|
||||
operationId: get_bank_profile
|
||||
@@ -1601,6 +1630,7 @@ paths:
|
||||
tags:
|
||||
- Banks
|
||||
put:
|
||||
deprecated: true
|
||||
description: "Update bank's disposition traits (skepticism, literalism, empathy)"
|
||||
operationId: update_bank_disposition
|
||||
parameters:
|
||||
@@ -1850,6 +1880,54 @@ paths:
|
||||
summary: Clear all observations
|
||||
tags:
|
||||
- Banks
|
||||
/v1/default/banks/{bank_id}/memories/{memory_id}/observations:
|
||||
delete:
|
||||
description: Delete all observations derived from a specific memory and reset
|
||||
it for re-consolidation. The memory itself is not deleted. A consolidation
|
||||
job is triggered automatically so the memory will produce fresh observations
|
||||
on the next consolidation run.
|
||||
operationId: clear_memory_observations
|
||||
parameters:
|
||||
- explode: false
|
||||
in: path
|
||||
name: bank_id
|
||||
required: true
|
||||
schema:
|
||||
title: Bank Id
|
||||
type: string
|
||||
style: simple
|
||||
- explode: false
|
||||
in: path
|
||||
name: memory_id
|
||||
required: true
|
||||
schema:
|
||||
title: Memory Id
|
||||
type: string
|
||||
style: simple
|
||||
- explode: false
|
||||
in: header
|
||||
name: authorization
|
||||
required: false
|
||||
schema:
|
||||
nullable: true
|
||||
type: string
|
||||
style: simple
|
||||
responses:
|
||||
"200":
|
||||
content:
|
||||
application/json:
|
||||
schema:
|
||||
$ref: '#/components/schemas/ClearMemoryObservationsResponse'
|
||||
description: Successful Response
|
||||
"422":
|
||||
content:
|
||||
application/json:
|
||||
schema:
|
||||
$ref: '#/components/schemas/HTTPValidationError'
|
||||
description: Validation Error
|
||||
summary: Clear observations for a memory
|
||||
tags:
|
||||
- Memory
|
||||
/v1/default/banks/{bank_id}/config:
|
||||
delete:
|
||||
description: Reset bank configuration to defaults by removing all bank-specific
|
||||
@@ -2595,6 +2673,17 @@ components:
|
||||
- created_at
|
||||
- document_id
|
||||
title: ChunkResponse
|
||||
ClearMemoryObservationsResponse:
|
||||
description: Response model for clearing observations for a specific memory.
|
||||
example:
|
||||
deleted_count: 3
|
||||
properties:
|
||||
deleted_count:
|
||||
title: Deleted Count
|
||||
type: integer
|
||||
required:
|
||||
- deleted_count
|
||||
title: ClearMemoryObservationsResponse
|
||||
ConsolidationResponse:
|
||||
description: Response model for consolidation trigger endpoint.
|
||||
example:
|
||||
@@ -2616,24 +2705,58 @@ components:
|
||||
CreateBankRequest:
|
||||
description: Request model for creating/updating a bank.
|
||||
example:
|
||||
disposition:
|
||||
empathy: 3
|
||||
literalism: 3
|
||||
skepticism: 3
|
||||
mission: I am a PM helping my engineering team stay organized
|
||||
name: Alice
|
||||
observations_mission: Observations are stable facts about people and projects.
|
||||
Always include preferences and skills.
|
||||
retain_mission: Always include technical decisions and architectural trade-offs.
|
||||
Ignore meeting logistics.
|
||||
properties:
|
||||
name:
|
||||
nullable: true
|
||||
type: string
|
||||
disposition:
|
||||
$ref: '#/components/schemas/DispositionTraits'
|
||||
disposition_skepticism:
|
||||
maximum: 5.0
|
||||
minimum: 1.0
|
||||
nullable: true
|
||||
type: integer
|
||||
disposition_literalism:
|
||||
maximum: 5.0
|
||||
minimum: 1.0
|
||||
nullable: true
|
||||
type: integer
|
||||
disposition_empathy:
|
||||
maximum: 5.0
|
||||
minimum: 1.0
|
||||
nullable: true
|
||||
type: integer
|
||||
mission:
|
||||
nullable: true
|
||||
type: string
|
||||
background:
|
||||
nullable: true
|
||||
type: string
|
||||
reflect_mission:
|
||||
nullable: true
|
||||
type: string
|
||||
retain_mission:
|
||||
nullable: true
|
||||
type: string
|
||||
retain_extraction_mode:
|
||||
nullable: true
|
||||
type: string
|
||||
retain_custom_instructions:
|
||||
nullable: true
|
||||
type: string
|
||||
retain_chunk_size:
|
||||
nullable: true
|
||||
type: integer
|
||||
enable_observations:
|
||||
nullable: true
|
||||
type: boolean
|
||||
observations_mission:
|
||||
nullable: true
|
||||
type: string
|
||||
title: CreateBankRequest
|
||||
CreateDirectiveRequest:
|
||||
description: Request model for creating a directive.
|
||||
@@ -3226,6 +3349,8 @@ components:
|
||||
$ref: '#/components/schemas/EntityIncludeOptions'
|
||||
chunks:
|
||||
$ref: '#/components/schemas/ChunkIncludeOptions'
|
||||
source_facts:
|
||||
$ref: '#/components/schemas/SourceFactsIncludeOptions'
|
||||
title: IncludeOptions
|
||||
ListDocumentsResponse:
|
||||
description: Response model for list documents endpoint.
|
||||
@@ -3352,9 +3477,7 @@ components:
|
||||
title: Content
|
||||
type: string
|
||||
timestamp:
|
||||
format: date-time
|
||||
nullable: true
|
||||
type: string
|
||||
$ref: '#/components/schemas/Timestamp'
|
||||
context:
|
||||
nullable: true
|
||||
type: string
|
||||
@@ -3375,6 +3498,8 @@ components:
|
||||
type: string
|
||||
nullable: true
|
||||
type: array
|
||||
observation_scopes:
|
||||
$ref: '#/components/schemas/ObservationScopes'
|
||||
required:
|
||||
- content
|
||||
title: MemoryItem
|
||||
@@ -3724,6 +3849,10 @@ components:
|
||||
additionalProperties:
|
||||
$ref: '#/components/schemas/ChunkData'
|
||||
nullable: true
|
||||
source_facts:
|
||||
additionalProperties:
|
||||
$ref: '#/components/schemas/RecallResult'
|
||||
nullable: true
|
||||
required:
|
||||
- results
|
||||
title: RecallResponse
|
||||
@@ -3789,6 +3918,11 @@ components:
|
||||
type: string
|
||||
nullable: true
|
||||
type: array
|
||||
source_fact_ids:
|
||||
items:
|
||||
type: string
|
||||
nullable: true
|
||||
type: array
|
||||
required:
|
||||
- id
|
||||
- text
|
||||
@@ -4083,9 +4217,6 @@ components:
|
||||
description: Request model for retain endpoint.
|
||||
example:
|
||||
async: false
|
||||
document_tags:
|
||||
- user_a
|
||||
- user_b
|
||||
items:
|
||||
- content: Alice works at Google
|
||||
context: work
|
||||
@@ -4148,6 +4279,15 @@ components:
|
||||
- items_count
|
||||
- success
|
||||
title: RetainResponse
|
||||
SourceFactsIncludeOptions:
|
||||
description: Options for including source facts for observation-type results.
|
||||
properties:
|
||||
max_tokens:
|
||||
default: 4096
|
||||
description: Maximum tokens for source facts
|
||||
title: Max Tokens
|
||||
type: integer
|
||||
title: SourceFactsIncludeOptions
|
||||
TagItem:
|
||||
description: Single tag with usage count.
|
||||
properties:
|
||||
@@ -4317,6 +4457,36 @@ components:
|
||||
- api_version
|
||||
- features
|
||||
title: VersionResponse
|
||||
Timestamp:
|
||||
anyOf:
|
||||
- format: date-time
|
||||
type: string
|
||||
- type: string
|
||||
description: "When the content occurred. Accepts an ISO 8601 datetime string\
|
||||
\ (e.g. '2024-01-15T10:30:00Z'), null/omitted (defaults to now), or the special\
|
||||
\ string 'unset' to explicitly store without any timestamp (use this for timeless\
|
||||
\ content such as fictional documents or static reference material)."
|
||||
nullable: true
|
||||
title: Timestamp
|
||||
ObservationScopes:
|
||||
anyOf:
|
||||
- enum:
|
||||
- per_tag
|
||||
- combined
|
||||
- all_combinations
|
||||
type: string
|
||||
- items:
|
||||
items:
|
||||
type: string
|
||||
type: array
|
||||
type: array
|
||||
description: "How to scope observations during consolidation. 'per_tag' runs\
|
||||
\ one consolidation pass per individual tag, creating separate observations\
|
||||
\ for each tag. 'combined' (default) runs a single pass with all tags together.\
|
||||
\ A list of tag lists runs one pass per inner list, giving full control over\
|
||||
\ which combinations to use."
|
||||
nullable: true
|
||||
title: ObservationScopes
|
||||
ValidationError_loc_inner:
|
||||
anyOf:
|
||||
- type: string
|
||||
|
||||
@@ -3,7 +3,7 @@ Hindsight HTTP API
|
||||
|
||||
HTTP API for Hindsight
|
||||
|
||||
API version: 0.4.11
|
||||
API version: 0.4.14
|
||||
*/
|
||||
|
||||
// Code generated by OpenAPI Generator (https://openapi-generator.tech); DO NOT EDIT.
|
||||
@@ -804,6 +804,8 @@ Get disposition traits and mission for a memory bank. Auto-creates agent with de
|
||||
@param ctx context.Context - for authentication, logging, cancellation, deadlines, tracing, etc. Passed from http.Request or context.Background().
|
||||
@param bankId
|
||||
@return ApiGetBankProfileRequest
|
||||
|
||||
Deprecated
|
||||
*/
|
||||
func (a *BanksAPIService) GetBankProfile(ctx context.Context, bankId string) ApiGetBankProfileRequest {
|
||||
return ApiGetBankProfileRequest{
|
||||
@@ -815,6 +817,7 @@ func (a *BanksAPIService) GetBankProfile(ctx context.Context, bankId string) Api
|
||||
|
||||
// Execute executes the request
|
||||
// @return BankProfileResponse
|
||||
// Deprecated
|
||||
func (a *BanksAPIService) GetBankProfileExecute(r ApiGetBankProfileRequest) (*BankProfileResponse, *http.Response, error) {
|
||||
var (
|
||||
localVarHTTPMethod = http.MethodGet
|
||||
@@ -1560,6 +1563,8 @@ Update bank's disposition traits (skepticism, literalism, empathy)
|
||||
@param ctx context.Context - for authentication, logging, cancellation, deadlines, tracing, etc. Passed from http.Request or context.Background().
|
||||
@param bankId
|
||||
@return ApiUpdateBankDispositionRequest
|
||||
|
||||
Deprecated
|
||||
*/
|
||||
func (a *BanksAPIService) UpdateBankDisposition(ctx context.Context, bankId string) ApiUpdateBankDispositionRequest {
|
||||
return ApiUpdateBankDispositionRequest{
|
||||
@@ -1571,6 +1576,7 @@ func (a *BanksAPIService) UpdateBankDisposition(ctx context.Context, bankId stri
|
||||
|
||||
// Execute executes the request
|
||||
// @return BankProfileResponse
|
||||
// Deprecated
|
||||
func (a *BanksAPIService) UpdateBankDispositionExecute(r ApiUpdateBankDispositionRequest) (*BankProfileResponse, *http.Response, error) {
|
||||
var (
|
||||
localVarHTTPMethod = http.MethodPut
|
||||
|
||||
@@ -3,7 +3,7 @@ Hindsight HTTP API
|
||||
|
||||
HTTP API for Hindsight
|
||||
|
||||
API version: 0.4.11
|
||||
API version: 0.4.14
|
||||
*/
|
||||
|
||||
// Code generated by OpenAPI Generator (https://openapi-generator.tech); DO NOT EDIT.
|
||||
|
||||
@@ -3,7 +3,7 @@ Hindsight HTTP API
|
||||
|
||||
HTTP API for Hindsight
|
||||
|
||||
API version: 0.4.11
|
||||
API version: 0.4.14
|
||||
*/
|
||||
|
||||
// Code generated by OpenAPI Generator (https://openapi-generator.tech); DO NOT EDIT.
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user