feat: support for gemini-3-pro and gpt-5.2 (#30)

* feat: support for gemini-3-pro and gpt-5.2

* feat: support for gemini-3-pro and gpt-5.2

* feat: support for gemini-3-pro and gpt-5.2

* feat: support for gemini-3-pro and gpt-5.2

* feat: add local mcp server

* docs

* docs
This commit is contained in:
Nicolò Boschi
2025-12-16 11:00:27 +01:00
committed by GitHub
parent 7dd68538bb
commit bb1f9cb221
11 changed files with 5617 additions and 123 deletions
+27
View File
@@ -0,0 +1,27 @@
#!/bin/bash
# Pre-commit hook - runs all scripts in scripts/hooks/
set -e
REPO_ROOT="$(git rev-parse --show-toplevel)"
HOOKS_DIR="$REPO_ROOT/scripts/hooks"
if [ ! -d "$HOOKS_DIR" ]; then
exit 0
fi
echo ""
echo "=== Running pre-commit hooks ==="
echo ""
# Run all executable scripts in hooks directory
for hook in "$HOOKS_DIR"/*.sh; do
if [ -x "$hook" ]; then
echo "[hook] $(basename "$hook")"
(cd "$REPO_ROOT" && "$hook")
fi
done
echo ""
echo "=== Pre-commit hooks completed ==="
echo ""
@@ -282,3 +282,81 @@ If no genuine opinions are expressed (e.g., the response just says "I don't know
except Exception as e:
logger.warning(f"Failed to extract opinions: {str(e)}")
return []
async def reflect(
llm_config,
query: str,
experience_facts: List[str] = None,
world_facts: List[str] = None,
opinion_facts: List[str] = None,
name: str = "Assistant",
disposition: DispositionTraits = None,
background: str = "",
context: str = None,
) -> str:
"""
Standalone reflect function for generating answers based on facts.
This is a static version of the reflect operation that can be called
without a MemoryEngine instance, useful for testing.
Args:
llm_config: LLM provider instance
query: Question to answer
experience_facts: List of experience/agent fact strings
world_facts: List of world fact strings
opinion_facts: List of opinion fact strings
name: Name of the agent/persona
disposition: Disposition traits (defaults to neutral)
background: Background information
context: Additional context for the prompt
Returns:
Generated answer text
"""
# Default disposition if not provided
if disposition is None:
disposition = DispositionTraits(skepticism=3, literalism=3, empathy=3)
# Convert string lists to MemoryFact format for formatting
def to_memory_facts(facts: List[str], fact_type: str) -> List[MemoryFact]:
if not facts:
return []
return [MemoryFact(id=f"test-{i}", text=f, fact_type=fact_type) for i, f in enumerate(facts)]
agent_results = to_memory_facts(experience_facts or [], "experience")
world_results = to_memory_facts(world_facts or [], "world")
opinion_results = to_memory_facts(opinion_facts or [], "opinion")
# Format facts for prompt
agent_facts_text = format_facts_for_prompt(agent_results)
world_facts_text = format_facts_for_prompt(world_results)
opinion_facts_text = format_facts_for_prompt(opinion_results)
# Build prompt
prompt = build_think_prompt(
agent_facts_text=agent_facts_text,
world_facts_text=world_facts_text,
opinion_facts_text=opinion_facts_text,
query=query,
name=name,
disposition=disposition,
background=background,
context=context,
)
system_message = get_system_message(disposition)
# Call LLM
answer_text = await llm_config.call(
messages=[
{"role": "system", "content": system_message},
{"role": "user", "content": prompt}
],
scope="memory_think",
temperature=0.9,
max_completion_tokens=1000
)
return answer_text.strip()
+51 -68
View File
@@ -1,9 +1,12 @@
"""
Test LLM provider with different models and providers.
Test LLM provider with different models using actual memory operations.
"""
import os
from datetime import datetime
import pytest
from hindsight_api.engine.llm_wrapper import LLMProvider
from hindsight_api.engine.utils import extract_facts
from hindsight_api.engine.search.think_utils import reflect
# Model matrix: (provider, model)
@@ -15,13 +18,14 @@ MODEL_MATRIX = [
("openai", "gpt-5-mini"),
("openai", "gpt-5-nano"),
("openai", "gpt-5"),
("openai", "gpt-5.2"),
# Groq models
("groq", "llama-3.3-70b-versatile"),
("groq", "openai/gpt-oss-120b"),
("groq", "openai/gpt-oss-20b"),
# Gemini models
("gemini", "gemini-2.5-flash"),
("gemini", "gemini-2.5-flash-lite"),
("gemini", "gemini-3-pro-preview"),
]
@@ -38,10 +42,10 @@ def get_api_key_for_provider(provider: str) -> str | None:
@pytest.mark.parametrize("provider,model", MODEL_MATRIX)
@pytest.mark.asyncio
async def test_llm_provider_call(provider: str, model: str):
async def test_llm_provider_memory_operations(provider: str, model: str):
"""
Test LLM provider can make a basic call with different models.
Skips if the required API key is not available.
Test LLM provider with actual memory operations: fact extraction and reflect.
All models must pass this test.
"""
api_key = get_api_key_for_provider(provider)
if not api_key:
@@ -54,74 +58,53 @@ async def test_llm_provider_call(provider: str, model: str):
model=model,
)
# Test basic call
response = await llm.call(
messages=[{"role": "user", "content": "Say 'hello' and nothing else."}],
max_completion_tokens=50,
temperature=0.1,
# Test 1: Fact extraction (structured output)
test_text = """
User: I just got back from my trip to Paris last week. The Eiffel Tower was amazing!
Assistant: That sounds wonderful! How long were you there?
User: About 5 days. I also visited the Louvre and saw the Mona Lisa.
"""
event_date = datetime(2024, 12, 10)
facts, chunks = await extract_facts(
text=test_text,
event_date=event_date,
context="Travel conversation",
llm_config=llm,
)
print(f"\n{provider}/{model} response: {response}")
assert response is not None, f"{provider}/{model} returned None"
print(f"\n{provider}/{model} - Fact extraction:")
print(f" Extracted {len(facts)} facts from {len(chunks)} chunks")
for fact in facts:
print(f" - {fact.fact}")
assert facts is not None, f"{provider}/{model} fact extraction returned None"
assert len(facts) > 0, f"{provider}/{model} should extract at least one fact"
@pytest.mark.parametrize("provider,model", MODEL_MATRIX)
@pytest.mark.asyncio
async def test_llm_provider_verify_connection(provider: str, model: str):
"""
Test LLM provider verify_connection method with different models.
Skips if the required API key is not available.
"""
api_key = get_api_key_for_provider(provider)
if not api_key:
pytest.skip(f"Skipping {provider}/{model}: no API key available")
# Verify facts have required fields
for fact in facts:
assert fact.fact, f"{provider}/{model} fact missing text"
assert fact.fact_type in ["world", "experience", "opinion"], f"{provider}/{model} invalid fact_type: {fact.fact_type}"
llm = LLMProvider(
provider=provider,
api_key=api_key,
base_url="",
model=model,
# Test 2: Reflect (actual reflect function)
response = await reflect(
llm_config=llm,
query="What was the highlight of my Paris trip?",
experience_facts=[
"I visited Paris in December 2024",
"I saw the Eiffel Tower and it was amazing",
"I visited the Louvre and saw the Mona Lisa",
"The trip lasted 5 days",
],
world_facts=[
"The Eiffel Tower is a famous landmark in Paris",
"The Mona Lisa is displayed at the Louvre museum",
],
name="Traveler",
)
# Test verify_connection
await llm.verify_connection()
print(f"\n{provider}/{model} connection verified")
print(f"\n{provider}/{model} - Reflect response:")
print(f" {response[:200]}...")
# Models that support large output (65000+ tokens)
LARGE_OUTPUT_MODELS = [
("openai", "gpt-5-mini"),
("openai", "gpt-5-nano"),
("openai", "gpt-5"),
("gemini", "gemini-2.5-flash"),
("gemini", "gemini-2.5-flash-lite"),
]
@pytest.mark.parametrize("provider,model", LARGE_OUTPUT_MODELS)
@pytest.mark.asyncio
async def test_llm_provider_large_output(provider: str, model: str):
"""
Test LLM provider with large max_completion_tokens (65000).
Only tests models that support large outputs.
Skips if the required API key is not available.
"""
api_key = get_api_key_for_provider(provider)
if not api_key:
pytest.skip(f"Skipping {provider}/{model}: no API key available")
llm = LLMProvider(
provider=provider,
api_key=api_key,
base_url="",
model=model,
)
# Test call with large max_completion_tokens
response = await llm.call(
messages=[{"role": "user", "content": "Say 'ok'"}],
max_completion_tokens=65000,
)
print(f"\n{provider}/{model} large output response: {response}")
assert response is not None, f"{provider}/{model} returned None"
assert response is not None, f"{provider}/{model} reflect returned None"
assert len(response) > 10, f"{provider}/{model} reflect response too short"
+2 -1
View File
@@ -26,17 +26,18 @@ The following models have been tested and verified to work correctly with Hindsi
| Provider | Model |
|----------|-------|
| **OpenAI** | `gpt-5.2` |
| **OpenAI** | `gpt-5` |
| **OpenAI** | `gpt-5-mini` |
| **OpenAI** | `gpt-5-nano` |
| **OpenAI** | `gpt-4.1-mini` |
| **OpenAI** | `gpt-4.1-nano` |
| **OpenAI** | `gpt-4o-mini` |
| **Gemini** | `gemini-3-pro-preview` |
| **Gemini** | `gemini-2.5-flash` |
| **Gemini** | `gemini-2.5-flash-lite` |
| **Groq** | `openai/gpt-oss-120b` |
| **Groq** | `openai/gpt-oss-20b` |
| **Groq** | `llama-3.3-70b-versatile` |
### Using Other Models
@@ -65,13 +65,13 @@ By default, memories are stored in a bank called `mcp`. To use a different bank:
## Environment Variables
All standard [Hindsight configuration variables](/developer/configuration) are supported.
### Local MCP Specific
| Variable | Required | Default | Description |
|----------|----------|---------|-------------|
| `HINDSIGHT_API_LLM_API_KEY` | Yes | - | API key for the LLM provider |
| `HINDSIGHT_API_LLM_PROVIDER` | No | `openai` | LLM provider (`openai`, `groq`, `anthropic`) |
| `HINDSIGHT_API_LLM_MODEL` | No | `gpt-4o-mini` | Model to use for fact extraction |
| `HINDSIGHT_API_MCP_LOCAL_BANK_ID` | No | `mcp` | Memory bank ID |
| `HINDSIGHT_API_LOG_LEVEL` | No | `info` | Log level (`debug`, `info`, `warning`, `error`) |
| `HINDSIGHT_API_MCP_LOCAL_BANK_ID` | No | `mcp` | Memory bank ID to use |
## Available Tools
@@ -125,23 +125,6 @@ Search memories to provide personalized responses.
}
```
**Response:**
```json
{
"results": [
{
"id": "...",
"text": "User's favorite color is blue",
"fact_type": "world",
"context": "preferences",
"event_date": null,
"score": 0.95
}
],
"total_tokens": 42
}
```
## How It Works
The local MCP server:
@@ -152,16 +135,6 @@ The local MCP server:
Data is persisted in the pg0 data directory (`~/.pg0/hindsight-mcp/`), so your memories survive restarts.
## Comparison: Local vs Server MCP
| Feature | Local MCP | Server MCP |
|---------|-----------|------------|
| Setup | Zero config | Requires running server |
| Database | Embedded (pg0) | External PostgreSQL |
| Multi-user | Single user | Multi-tenant |
| Scalability | Single machine | Horizontally scalable |
| Use case | Personal/development | Production/teams |
## Troubleshooting
### "HINDSIGHT_API_LLM_API_KEY required"
+5
View File
@@ -152,6 +152,11 @@ const sidebars: SidebarsConfig = {
label: 'Integrations',
collapsible: false,
items: [
{
type: 'doc',
id: 'sdks/integrations/local-mcp',
label: 'Local MCP Server',
},
{
type: 'doc',
id: 'sdks/integrations/litellm',
+11 -21
View File
@@ -509,32 +509,22 @@ article a:not(.button):not([class*="hash-link"]) {
font-weight: 500;
}
/* Links containing code - the code inherits the transparent text-fill from the link */
article a code {
-webkit-text-fill-color: #3396e8 !important;
color: #3396e8 !important;
}
article a:hover code {
-webkit-text-fill-color: #0074d9 !important;
color: #0074d9 !important;
}
article a:not(.button):not([class*="hash-link"]):hover {
text-decoration: underline;
text-decoration-color: var(--hindsight-gradient-start);
}
/* Links inside code blocks - use solid color instead of gradient */
code a,
pre a,
article code a,
article pre a {
background: none !important;
-webkit-background-clip: unset !important;
-webkit-text-fill-color: var(--ifm-color-primary) !important;
background-clip: unset !important;
color: var(--ifm-color-primary) !important;
text-decoration: underline;
}
code a:hover,
pre a:hover,
article code a:hover,
article pre a:hover {
color: var(--ifm-color-primary-dark) !important;
-webkit-text-fill-color: var(--ifm-color-primary-dark) !important;
}
/* Admonitions - gradient themed */
.theme-admonition,
[class*="admonition_"] {
File diff suppressed because it is too large Load Diff
+4 -1
View File
@@ -5,5 +5,8 @@
"hindsight-clients/typescript",
"hindsight-control-plane",
"hindsight-docs"
]
],
"scripts": {
"prepare": "./scripts/setup-hooks.sh"
}
}
+35
View File
@@ -0,0 +1,35 @@
#!/bin/bash
# Regenerate llms-full.txt when docs change
LOG_PREFIX=" "
# Check if any docs files are staged
DOCS_CHANGED=$(git diff --cached --name-only -- 'hindsight-docs/docs/**/*.md' 'hindsight-docs/docs/**/*.mdx' 2>/dev/null || true)
if [ -z "$DOCS_CHANGED" ]; then
echo "${LOG_PREFIX}No docs changes, skipping"
exit 0
fi
echo "${LOG_PREFIX}Docs changed, regenerating llms-full.txt..."
# Check if npm is available and hindsight-docs exists
if [ ! -d "hindsight-docs" ] || ! command -v npm &> /dev/null; then
echo "${LOG_PREFIX}Warning: Cannot regenerate llms-full.txt (missing hindsight-docs or npm)"
exit 0
fi
cd hindsight-docs
# Run the generate script
if npm run generate-llms --silent 2>/dev/null; then
# Check if llms-full.txt changed
if [ -n "$(git diff --name-only -- static/llms-full.txt 2>/dev/null)" ]; then
echo "${LOG_PREFIX}llms-full.txt updated, staging..."
git add static/llms-full.txt
else
echo "${LOG_PREFIX}llms-full.txt unchanged"
fi
else
echo "${LOG_PREFIX}Warning: Failed to regenerate llms-full.txt"
fi
+11
View File
@@ -0,0 +1,11 @@
#!/bin/bash
# Setup git hooks for the repository
set -e
REPO_ROOT="$(git rev-parse --show-toplevel)"
echo "Setting up git hooks..."
git config core.hooksPath "$REPO_ROOT/.githooks"
echo "Git hooks configured to use .githooks directory"
echo "Done!"