perf think improvements

This commit is contained in:
Nicolò Boschi 2025-11-05 10:14:47 +01:00
parent ba048ff96c
commit d8fdbb94ef
10 changed files with 1227 additions and 0 deletions

View file

@ -0,0 +1,356 @@
{
"overall_accuracy": 100.0,
"total_correct": 1,
"total_questions": 1,
"num_items": 1,
"item_results": [
{
"item_id": "conv-26",
"metrics": {
"accuracy": 100.0,
"correct": 1,
"total": 1,
"category_stats": {
"2": {
"correct": 1,
"total": 1
}
},
"detailed_results": [
{
"question": "When did Caroline go to the LGBTQ support group?",
"correct_answer": "7 May 2023",
"predicted_answer": "Caroline attended the LGBTQ support group on May\u202f7\u202f2023 (event recorded at 2023\u201105\u201107T13:56:00\u202fUTC).",
"reasoning": "Think API: 20 world facts, 0 agent facts, 20 opinions",
"category": 2,
"retrieved_memories": [
{
"id": "5ddabf72-967c-4bd0-a0a8-1545324e48f2",
"text": "Caroline attended an LGBTQ support group on 2023-05-07 and found it powerful.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_1)",
"event_date": "2023-05-07T13:56:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "7b85f829-743e-4827-8349-ca6d01dba5dc",
"text": "Caroline volunteered at an LGBTQ+ youth center",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_15)",
"event_date": "2023-08-28T15:19:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "6794d0e8-5e4d-44bb-8238-21249021d101",
"text": "Caroline joined the LGBTQ activist group \"Connected LGBTQ Activists\" on 2023-07-18.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_10)",
"event_date": "2023-07-18T20:56:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "eb8952a1-9b7d-45b3-8f63-2411d6421725",
"text": "Caroline tried to apologize to the people she encountered during the hike.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_14)",
"event_date": "2023-08-18T13:33:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "2decc740-9674-4b25-acf1-1da8ba8f61dd",
"text": "Caroline owns a guinea pig named Oscar.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_13)",
"event_date": "2023-08-23T15:31:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "1aad5731-397b-4c3f-b78d-58768d376133",
"text": "Caroline told Melanie that she is lucky to have such an awesome family.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_10)",
"event_date": "2023-07-20T20:56:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "5b3e87cc-6892-4fb2-945e-5921a32c3453",
"text": "Caroline attended a poetry reading on Friday, October 6, 2023.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_17)",
"event_date": "2023-10-06T10:31:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "3fd7f709-6b78-443c-8768-abeee8f28e31",
"text": "Caroline created a self-portrait last week and posted a photo of it.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_13)",
"event_date": "2023-08-16T15:31:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "640e4293-755b-4634-b570-1f15128eb1e6",
"text": "The event room was electric with energy and support, and the posters displayed pride and strength, which inspired Caroline to create new artwork.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_17)",
"event_date": "2023-10-06T10:31:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "a9331948-bf5f-4791-86e9-0bf62342ed88",
"text": "Caroline visited the beach.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_14)",
"event_date": "2023-08-18T13:33:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "c48b29f0-bcb4-4fde-b616-4e08d9166146",
"text": "Caroline attended an adoption advice and assistance group, receiving a lot of help.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_13)",
"event_date": "2023-08-23T15:31:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "eba11741-e2ae-4ccf-976a-4df864b0cace",
"text": "Caroline heard transgender stories at the LGBTQ support group, which she found inspiring.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_1)",
"event_date": "2023-05-07T13:56:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "5968eee3-380d-4780-afec-9576fbafe422",
"text": "Caroline is interested in a career in counseling or mental health to support people with similar issues.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_1)",
"event_date": "2023-05-08T13:56:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "aee0214f-e49d-47a4-83f5-5f97304a3a51",
"text": "Caroline feels supported by people around her, which makes her feel okay.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_12)",
"event_date": "2023-08-17T13:50:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "149347b4-b7cb-4bf9-a8be-f743700a9783",
"text": "The city held a pride parade on 2023-07-15, where many people marched, waved flags, held signs, and celebrated love and diversity; Caroline missed the parade.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_10)",
"event_date": "2023-07-15T20:56:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "21ac438b-ce0b-4efd-8c6e-d424295ec3cc",
"text": "Caroline created a recent painting that represents inclusivity and diversity and uses it to speak up for the LGBTQ+ community and push for acceptance.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_11)",
"event_date": "2023-08-14T14:24:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "442b4482-f31b-4ef5-b545-eb0befab7cfe",
"text": "Caroline mentioned an advocacy event that was a cool experience with love and support.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_11)",
"event_date": "2023-08-14T14:24:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "a100f54b-7112-4616-8f7c-9a7347587597",
"text": "Caroline had a not-so-great experience on a hike where she ran into a group of religious conservatives who said something that upset her, leading her to reflect on the need for more work on LGBTQ rights.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_12)",
"event_date": "2023-08-17T13:50:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "3e2e6418-29db-44e4-963f-ea2089d33dd5",
"text": "Caroline has a lifelong love for horses.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_13)",
"event_date": "2023-08-23T15:31:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "0dda577c-9ea3-4a47-835a-63cade830838",
"text": "Caroline gave a talk at her school event last week about her transgender journey, encouraging students to get involved in the LGBTQ community and observed positive reactions from the audience.",
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_3)",
"event_date": "2023-06-02T19:55:00+00:00",
"score": 0.0,
"fact_type": "world"
},
{
"id": "054d8d37-aa7d-4a21-b4bd-7968e29d74dd",
"text": "Caroline is a participant within the LGBTQ community",
"context": "formed during thinking about: In what ways is Caroline participating in the LGBTQ community?",
"event_date": "2025-11-04T18:24:51+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "9862a174-0e48-418c-b796-2cb1185b7f61",
"text": "Caroline's plan to build a personal library for her future children suggests she may consider formal study in library science or related fields.",
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
"event_date": "2025-11-04T15:30:15+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "7318bff4-0702-44b9-8aa6-b50f5a59be86",
"text": "We cannot determine whether Caroline wants to move back to her home country soon",
"context": "formed during thinking about: Would Caroline want to move back to her home country soon?",
"event_date": "2025-11-04T18:05:51+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "cdb11bda-7c70-4b61-b36e-bdf21727a194",
"text": "Caroline is an event organizer within the LGBTQ community",
"context": "formed during thinking about: In what ways is Caroline participating in the LGBTQ community?",
"event_date": "2025-11-04T18:24:51+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "1d68207b-03e2-44ca-a79b-47ba427c1ba1",
"text": "Melanie and Caroline describe their journey as collaborative, supportive, and inspiring",
"context": "formed during thinking about: How do Melanie and Caroline describe their journey through life together?",
"event_date": "2025-11-04T18:24:08+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "8b6a13ff-7417-4458-a19c-cd101d24c53f",
"text": "Caroline has passed adoption agency interviews.",
"context": "formed during thinking about: What is Caroline's identity?",
"event_date": "2025-11-04T15:30:19+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "4e8a04ef-ea96-48ec-b016-c32a111bc819",
"text": "Caroline is an advocate within the LGBTQ community",
"context": "formed during thinking about: In what ways is Caroline participating in the LGBTQ community?",
"event_date": "2025-11-04T18:24:51+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "87c97473-6596-4e5c-bd0f-f989ab76e137",
"text": "The provided facts only describe Caroline's activities, goals, and interests in her current location",
"context": "formed during thinking about: Would Caroline want to move back to her home country soon?",
"event_date": "2025-11-04T18:05:51+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "6aa8c0fd-dffc-4d27-9559-b0687d263f59",
"text": "Caroline plays piano.",
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
"event_date": "2025-11-04T15:27:06+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "04372c16-57ae-429b-8f9c-e3ab9ef3bacd",
"text": "Caroline could integrate art therapy classes into counseling work.",
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
"event_date": "2025-11-04T15:27:06+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "d596524c-fdd6-4080-b7a7-f9da2cd3d6ed",
"text": "Caroline explicitly said she wants to continue her education.",
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
"event_date": "2025-11-04T18:08:28+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "f5308bcf-a59e-4b1d-90e0-0e243b79b98d",
"text": "Caroline paints.",
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
"event_date": "2025-11-04T15:27:06+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "f97a5f20-1e8d-45bc-93db-6626070ca5f1",
"text": "Caroline is interested in a career in mental-health services.",
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
"event_date": "2025-11-04T18:08:28+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "fc805a93-3b1d-4f8a-a98e-3b7044bcb2d0",
"text": "Caroline feels a sense of empowerment from activism.",
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
"event_date": "2025-11-04T15:27:06+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "31c26b9a-f4a3-4a08-bae8-410780b40414",
"text": "Previously I thought Caroline would likely focus on LGBTQ+ studies, gender studies, or social justice based on her activism. However, her explicit statements about pursuing a career in counseling and mental health, and her desire to support people with similar issues, indicate that her primary focus is now likely on a counseling or mental\u2011health career, potentially integrating her advocacy for marginalized communities.",
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
"event_date": "2025-11-04T15:27:06+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "987325a5-6b67-42bf-8910-fd194987bd35",
"text": "Caroline is a future parent-to-be.",
"context": "formed during thinking about: What is Caroline's identity?",
"event_date": "2025-11-04T15:30:19+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "ba5ae772-2d00-4cbf-ac88-6988bca1be78",
"text": "Melanie and Caroline frame challenges as a shared roller-coaster",
"context": "formed during thinking about: How do Melanie and Caroline describe their journey through life together?",
"event_date": "2025-11-04T18:24:08+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "de3b98e6-f435-41b7-a38b-712a35cd0c54",
"text": "There is no information indicating Caroline identifies as religious.",
"context": "formed during thinking about: Would Caroline be considered religious?",
"event_date": "2025-11-04T18:05:26+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "314d9820-e75d-455f-bfad-55dcea90c044",
"text": "Caroline may choose a multidisciplinary program that combines counseling/mental\u2011health training, LGBTQ+ advocacy, and early childhood literacy.",
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
"event_date": "2025-11-04T15:30:15+00:00",
"score": 0.0,
"fact_type": "opinion"
},
{
"id": "5ba2cc08-4998-4667-8b95-adb1f0986504",
"text": "Caroline is most likely to pursue LGBTQ+ Studies and Gender & Social Justice.",
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
"event_date": "2025-11-04T15:30:15+00:00",
"score": 0.0,
"fact_type": "opinion"
}
],
"is_correct": true,
"correctness_reasoning": "The generated answer states the same date (May\u202f7\u202f2023) as the gold answer, so it is correct."
}
]
},
"num_sessions": -1
}
]
}

View file

@ -0,0 +1,7 @@
# LoComo Benchmark Results (Think Mode)
**Overall Accuracy**: 100.00% (1/1)
| Sample ID | Sessions | Questions | Correct | Accuracy | Multi-hop | Single-hop | Temporal | Open-domain |
|-----------|----------|-----------|---------|----------|-----------|------------|----------|-------------|
| conv-26 | -1 | 1 | 1 | 100.00% | N/A | N/A | N/A | N/A |

View file

@ -0,0 +1,62 @@
# Benchmark Visualizer
A standalone web service for visualizing benchmark results. Currently supports the LoComo benchmark with plans to add more benchmarks in the future.
## Features
- Interactive web interface for viewing benchmark results
- Detailed breakdown by category (Multi-hop, Single-hop, Temporal, Open-domain)
- Filter options to view all, correct, or incorrect answers
- Expandable Q&A details with reasoning and retrieved memories
- Overall and per-item accuracy statistics
## Running the Visualizer
### Option 1: Using the serve script (recommended)
```bash
cd benchmarks/visualizer
./serve.sh
```
### Option 2: Using uvicorn directly
```bash
cd benchmarks/visualizer
uv run uvicorn server:app --reload --host 0.0.0.0 --port 8001
```
Then open your browser to: http://localhost:8001
## Usage
1. Select a benchmark from the dropdown:
- **LoComo (search)**: Traditional two-step approach (search → LLM answer generation)
- **LoComo (think)**: Integrated approach using think API (single call for retrieval + reasoning)
2. The visualization will automatically load and display:
- Overall accuracy statistics
- Category-wise performance breakdown
- Detailed results for each conversation
3. Use the filter controls to show all answers, only incorrect, or only correct answers
4. Expand individual conversations to see Q&A details, reasoning, and retrieved memories
## API Endpoints
- `GET /` - Main visualizer page
- `GET /api/locomo?mode={search|think}` - Returns LoComo benchmark results as JSON
- `mode=search` (default): Returns results from `benchmark_results.json`
- `mode=think`: Returns results from `benchmark_results_think.json`
## Requirements
- FastAPI
- Uvicorn
- Python 3.11+
The visualizer reads benchmark results from:
- `benchmarks/locomo/benchmark_results.json` for search mode
- `benchmarks/locomo/benchmark_results_think.json` for think mode
Make sure to run the benchmark first to generate results:
- Search mode: `cd benchmarks/locomo && uv run python run_benchmark.py`
- Think mode: `cd benchmarks/locomo && uv run python run_benchmark.py --use-think`

4
benchmarks/visualizer/serve.sh Executable file
View file

@ -0,0 +1,4 @@
#!/bin/bash
# Start the Benchmark Visualizer server with hot reload
cd "$(dirname "$0")"
uv run uvicorn server:app --reload --host 0.0.0.0 --port 8001

View file

@ -0,0 +1,72 @@
"""Benchmark Visualizer Web Service.
A standalone web service for visualizing benchmark results.
Currently supports LoComo benchmark visualization.
"""
import json
from pathlib import Path
from typing import Any
from fastapi import FastAPI, HTTPException
from fastapi.responses import HTMLResponse, FileResponse
from fastapi.staticfiles import StaticFiles
app = FastAPI(title="Benchmark Visualizer")
# Get the benchmarks directory
BENCHMARKS_DIR = Path(__file__).parent.parent
@app.get("/", response_class=HTMLResponse)
async def index():
"""Serve the main benchmark visualizer page."""
html_path = Path(__file__).parent / "static" / "index.html"
with open(html_path) as f:
return f.read()
@app.get("/api/locomo")
async def get_locomo_results(mode: str = "search") -> dict[str, Any]:
"""Get LoComo benchmark results.
Returns pre-computed benchmark results from the locomo directory.
Args:
mode: Either "search" (default) or "think" to select which results to load
"""
try:
# Determine filename based on mode
if mode == "think":
filename = "benchmark_results_think.json"
else:
filename = "benchmark_results.json"
results_path = BENCHMARKS_DIR / "locomo" / filename
if not results_path.exists():
raise HTTPException(
status_code=404,
detail=f"Benchmark results not found for mode '{mode}'. Please run the benchmark first with {'--use-think' if mode == 'think' else 'default settings'}."
)
with open(results_path) as f:
results = json.load(f)
return results
except json.JSONDecodeError as e:
raise HTTPException(
status_code=500,
detail=f"Failed to parse benchmark results: {str(e)}"
)
except Exception as e:
raise HTTPException(status_code=500, detail=str(e))
# Mount static files
app.mount("/static", StaticFiles(directory=Path(__file__).parent / "static"), name="static")
if __name__ == "__main__":
import uvicorn
uvicorn.run(app, host="127.0.0.1", port=8001)

View file

@ -0,0 +1,140 @@
body {
font-family: Tahoma, sans-serif;
margin: 0;
padding: 0;
background: #f5f5f5;
}
.header {
background: #333;
color: white;
padding: 20px;
border-bottom: 3px solid #42a5f5;
}
.header h1 {
margin: 0 0 5px 0;
}
.header p {
margin: 0;
color: #ccc;
}
.benchmark-selector {
background: #f0f0f0;
padding: 15px 20px;
border-bottom: 2px solid #333;
display: flex;
gap: 15px;
align-items: center;
}
.benchmark-selector label {
font-weight: bold;
font-size: 14px;
}
.benchmark-selector select {
padding: 8px 12px;
border: 2px solid #42a5f5;
border-radius: 4px;
background: white;
color: #333;
font-size: 14px;
font-weight: bold;
cursor: pointer;
min-width: 200px;
}
#benchmark-content {
padding: 20px;
}
.welcome-message {
text-align: center;
padding: 60px 20px;
color: #666;
}
.welcome-message h2 {
color: #333;
}
.load-button {
padding: 8px 20px;
background: #66bb6a;
color: white;
border: none;
border-radius: 4px;
cursor: pointer;
font-weight: bold;
font-size: 14px;
margin-bottom: 20px;
}
.load-button:hover {
background: #43a047;
}
.error-message {
color: #d32f2f;
padding: 20px;
background: #ffebee;
border: 2px solid #ef5350;
border-radius: 8px;
margin: 20px;
max-width: 800px;
}
.error-message h3 {
margin-top: 0;
color: #c62828;
}
.error-message pre {
background: #f5f5f5;
padding: 10px;
border-radius: 4px;
overflow-x: auto;
color: #333;
font-family: monospace;
font-size: 13px;
}
.stats-grid {
display: grid;
grid-template-columns: repeat(auto-fit, minmax(200px, 1fr));
gap: 10px;
margin-top: 10px;
}
.stat-item {
padding: 12px;
background: white;
border: 1px solid #ddd;
border-radius: 4px;
text-align: center;
}
.stat-label {
font-weight: bold;
color: #666;
font-size: 12px;
margin-bottom: 5px;
}
.stat-value {
font-size: 24px;
color: #333;
font-weight: bold;
}
.qa-results {
margin-top: 20px;
}
.qa-item {
margin-bottom: 15px;
border-radius: 8px;
}

View file

@ -0,0 +1,32 @@
<!DOCTYPE html>
<html>
<head>
<title>Benchmark Visualizer</title>
<meta charset="utf-8">
<link rel="stylesheet" href="/static/css/styles.css">
</head>
<body>
<div class="header">
<h1>Benchmark Visualizer</h1>
<p>Analyze and visualize benchmark results</p>
</div>
<div class="benchmark-selector">
<label>Select Benchmark:</label>
<select id="benchmark-select" onchange="selectBenchmark()">
<option value="">-- Select a benchmark --</option>
<option value="locomo-search">LoComo (search)</option>
<option value="locomo-think">LoComo (think)</option>
</select>
</div>
<div id="benchmark-content">
<div class="welcome-message">
<h2>Welcome to Benchmark Visualizer</h2>
<p>Select a benchmark from the dropdown above to view results.</p>
</div>
</div>
<script src="/static/js/app.js"></script>
</body>
</html>

View file

@ -0,0 +1,342 @@
// Benchmark Visualizer App
let currentBenchmark = null;
let benchmarkData = null;
function selectBenchmark() {
const select = document.getElementById('benchmark-select');
currentBenchmark = select.value;
if (!currentBenchmark) {
document.getElementById('benchmark-content').innerHTML = `
<div class="welcome-message">
<h2>Welcome to Benchmark Visualizer</h2>
<p>Select a benchmark from the dropdown above to view results.</p>
</div>
`;
return;
}
// Load the selected benchmark
if (currentBenchmark === 'locomo-search') {
loadLocomoResults('search');
} else if (currentBenchmark === 'locomo-think') {
loadLocomoResults('think');
}
}
async function loadLocomoResults(mode = 'search') {
try {
const response = await fetch(`/api/locomo?mode=${mode}`);
if (!response.ok) {
const errorData = await response.json();
const modeLabel = mode === 'think' ? 'think' : 'search';
const runCommand = mode === 'think'
? 'uv run python run_benchmark.py --use-think'
: 'uv run python run_benchmark.py';
document.getElementById('benchmark-content').innerHTML = `
<div class="error-message">
<h3> Benchmark Results Not Found</h3>
<p>${errorData.detail || 'The requested benchmark results are not available.'}</p>
<p><strong>To generate ${modeLabel} mode results:</strong></p>
<pre style="background: #f5f5f5; padding: 10px; border-radius: 4px; overflow-x: auto;">cd benchmarks/locomo
${runCommand}</pre>
<p style="margin-top: 15px; font-size: 14px; color: #666;">
Once the benchmark completes, refresh this page and select "${mode === 'think' ? 'LoComo (think)' : 'LoComo (search)'}" again.
</p>
</div>
`;
return;
}
benchmarkData = await response.json();
console.log(`Loaded locomo data (${mode} mode):`, benchmarkData);
renderLocomoResults(mode);
} catch (e) {
console.error('Error loading benchmark results:', e);
document.getElementById('benchmark-content').innerHTML = `
<div class="error-message">
<h3> Error Loading Results</h3>
<p>${e.message}</p>
<p style="font-size: 12px; color: #666; margin-top: 10px;">Check the browser console for more details.</p>
</div>
`;
}
}
function renderLocomoResults(mode = 'search') {
if (!benchmarkData) return;
const content = document.getElementById('benchmark-content');
try {
// Handle both old and new structure
const results = benchmarkData.item_results || benchmarkData.conversation_results || [];
const numItems = benchmarkData.num_items || results.length;
console.log('Rendering results:', { resultsCount: results.length, numItems });
// Calculate per-category statistics
const categoryStats = {
1: { name: 'Multi-hop', correct: 0, total: 0 },
2: { name: 'Single-hop', correct: 0, total: 0 },
3: { name: 'Temporal', correct: 0, total: 0 },
4: { name: 'Open-domain', correct: 0, total: 0 }
};
// Aggregate across all items
results.forEach(item => {
if (item.metrics && item.metrics.detailed_results) {
item.metrics.detailed_results.forEach(result => {
const category = result.category;
if (categoryStats[category]) {
categoryStats[category].total++;
if (result.is_correct) {
categoryStats[category].correct++;
}
}
});
}
});
// Determine title based on mode
const modeLabel = mode === 'think' ? ' (Think Mode)' : ' (Search Mode)';
// Overall stats
const overallHtml = `
<div style="background: #f9f9f9; padding: 20px; border: 2px solid #333; border-radius: 8px; margin-bottom: 20px;">
<h3 style="margin-top: 0;">LoComo Benchmark${modeLabel} - Overall Performance</h3>
<div class="stats-grid">
<div class="stat-item">
<div class="stat-label">Overall Accuracy</div>
<div class="stat-value">${benchmarkData.overall_accuracy.toFixed(2)}%</div>
</div>
<div class="stat-item">
<div class="stat-label">Correct Answers</div>
<div class="stat-value">${benchmarkData.total_correct} / ${benchmarkData.total_questions}</div>
</div>
<div class="stat-item">
<div class="stat-label">Items</div>
<div class="stat-value">${numItems}</div>
</div>
</div>
<h4 style="margin: 20px 0 10px 0; padding-top: 15px; border-top: 1px solid #ddd;">Accuracy by Category</h4>
<div class="stats-grid" style="grid-template-columns: repeat(auto-fit, minmax(200px, 1fr));">
${Object.values(categoryStats).map(cat => {
const accuracy = cat.total > 0 ? ((cat.correct / cat.total) * 100).toFixed(1) : 0;
const color = accuracy >= 70 ? '#43a047' : accuracy >= 50 ? '#ff9800' : '#e53935';
return `
<div class="stat-item">
<div class="stat-label">${cat.name}</div>
<div class="stat-value" style="color: ${color};">${accuracy}%</div>
<div style="font-size: 11px; color: #666; margin-top: 4px;">${cat.correct} / ${cat.total}</div>
</div>
`;
}).join('')}
</div>
</div>
`;
// Filter controls
const filterHtml = `
<div style="margin-bottom: 20px; display: flex; gap: 10px; align-items: center;">
<label style="font-weight: bold;">Show:</label>
<label><input type="radio" name="answer-filter" value="all" checked onchange="filterAnswers()"> All Answers</label>
<label><input type="radio" name="answer-filter" value="incorrect" onchange="filterAnswers()"> Incorrect Only</label>
<label><input type="radio" name="answer-filter" value="correct" onchange="filterAnswers()"> Correct Only</label>
</div>
`;
// Build item sections
let itemsHtml = '';
results.forEach((item, idx) => {
const itemId = item.item_id || item.sample_id || `item-${idx}`;
const accuracy = item.metrics.accuracy.toFixed(2);
const correctCount = item.metrics.correct;
const totalCount = item.metrics.total;
itemsHtml += `
<div style="margin-bottom: 30px; border: 2px solid #333; border-radius: 8px; overflow: hidden;">
<div style="background: #f0f0f0; padding: 15px; border-bottom: 2px solid #333; cursor: pointer;" onclick="toggleConversation(${idx})">
<h3 style="margin: 0; display: flex; justify-content: space-between; align-items: center;">
<span>📊 ${itemId}</span>
<span style="font-size: 18px; color: ${accuracy >= 70 ? '#43a047' : accuracy >= 50 ? '#ff9800' : '#e53935'};">
${accuracy}% (${correctCount}/${totalCount})
</span>
</h3>
</div>
<div id="conv-${idx}" style="display: none; padding: 20px;">
${renderConversationDetails(item)}
</div>
</div>
`;
});
content.innerHTML = overallHtml + filterHtml + itemsHtml;
} catch (e) {
console.error('Error rendering Locomo results:', e);
content.innerHTML = `
<div class="error-message">
<strong>Error rendering results:</strong> ${e.message}<br>
<pre style="margin-top: 10px; font-size: 11px; overflow: auto;">${e.stack}</pre>
</div>
`;
}
}
function renderConversationDetails(conv) {
if (!conv || !conv.metrics) {
return '<div style="padding: 20px; color: #666;">No metrics available</div>';
}
const results = conv.metrics.detailed_results;
if (!results || !Array.isArray(results) || results.length === 0) {
return '<div style="padding: 20px; color: #666;">No detailed results available</div>';
}
let html = '<div class="qa-results">';
results.forEach((result, idx) => {
const isCorrect = result.is_correct;
const bgColor = isCorrect ? '#e8f5e9' : '#ffebee';
const icon = isCorrect ? '✅' : '❌';
const category = getCategoryName(result.category);
html += `
<div class="qa-item" data-correct="${isCorrect}" style="background: ${bgColor}; padding: 15px; margin-bottom: 15px; border: 1px solid #ddd; border-radius: 8px;">
<div style="display: flex; justify-content: space-between; align-items: flex-start; margin-bottom: 10px;">
<div style="flex: 1;">
<div style="font-weight: bold; font-size: 16px; margin-bottom: 8px;">
${icon} Question ${idx + 1} <span style="font-size: 12px; background: #666; color: white; padding: 2px 8px; border-radius: 4px; margin-left: 8px;">${category}</span>
</div>
<div style="margin-bottom: 8px;">
<b>Q:</b> ${result.question}
</div>
</div>
</div>
<div style="display: grid; grid-template-columns: 1fr 1fr; gap: 15px; margin-bottom: 10px;">
<div>
<div style="font-weight: bold; color: #43a047; margin-bottom: 4px;"> Correct Answer:</div>
<div style="background: white; padding: 8px; border-radius: 4px; border: 1px solid #ccc;">
${result.correct_answer}
</div>
</div>
<div>
<div style="font-weight: bold; color: ${isCorrect ? '#43a047' : '#e53935'}; margin-bottom: 4px;">
${isCorrect ? '✓' : '✗'} Predicted Answer:
</div>
<div style="background: white; padding: 8px; border-radius: 4px; border: 1px solid #ccc;">
${result.predicted_answer}
</div>
</div>
</div>
<details style="margin-top: 10px;">
<summary style="cursor: pointer; font-weight: bold; padding: 5px; background: rgba(255,255,255,0.5); border-radius: 4px;">
📝 Show Reasoning & Retrieved Memories
</summary>
<div style="margin-top: 10px; padding: 10px; background: white; border-radius: 4px;">
<div style="margin-bottom: 10px;">
<b>System Reasoning:</b>
<div style="padding: 8px; background: #f5f5f5; border-radius: 4px; margin-top: 4px;">
${result.reasoning}
</div>
</div>
<div style="margin-bottom: 10px;">
<b>Judge Reasoning:</b>
<div style="padding: 8px; background: #f5f5f5; border-radius: 4px; margin-top: 4px;">
${result.correctness_reasoning || 'N/A'}
</div>
</div>
<div>
<b>Retrieved Memories (${result.retrieved_memories ? result.retrieved_memories.length : 0}):</b>
${renderRetrievedMemories(result.retrieved_memories)}
</div>
</div>
</details>
</div>
`;
});
html += '</div>';
return html;
}
function renderRetrievedMemories(memories) {
if (!memories || !Array.isArray(memories) || memories.length === 0) {
return '<div style="padding: 8px; color: #999;">No memories retrieved</div>';
}
let html = '<div style="margin-top: 8px;">';
memories.forEach((mem, idx) => {
if (!mem) return;
const eventDate = mem.event_date ? new Date(mem.event_date).toLocaleString() : 'N/A';
// Determine border color based on fact type
let borderColor = '#42a5f5'; // default blue
let factTypeLabel = '';
if (mem.fact_type) {
factTypeLabel = `<span style="background: #666; color: white; padding: 2px 6px; border-radius: 3px; font-size: 10px; margin-left: 8px;">${mem.fact_type.toUpperCase()}</span>`;
if (mem.fact_type === 'world') {
borderColor = '#4caf50'; // green
} else if (mem.fact_type === 'agent') {
borderColor = '#ff9800'; // orange
} else if (mem.fact_type === 'opinion') {
borderColor = '#9c27b0'; // purple
}
}
html += `
<div style="padding: 8px; background: #f5f5f5; border-left: 3px solid ${borderColor}; margin-bottom: 8px;">
<div style="font-size: 11px; color: #666; margin-bottom: 4px;">
Rank #${idx + 1} | Score: ${mem.score ? mem.score.toFixed(4) : 'N/A'} | Event Date: ${eventDate}${factTypeLabel}
</div>
<div style="font-size: 13px;">${mem.text}</div>
</div>
`;
});
html += '</div>';
return html;
}
function getCategoryName(category) {
const categories = {
1: 'Multi-hop',
2: 'Single-hop',
3: 'Temporal',
4: 'Open-domain'
};
return categories[category] || 'Unknown';
}
function toggleConversation(idx) {
const elem = document.getElementById(`conv-${idx}`);
if (elem.style.display === 'none') {
elem.style.display = 'block';
} else {
elem.style.display = 'none';
}
}
function filterAnswers() {
const filter = document.querySelector('input[name="answer-filter"]:checked').value;
const items = document.querySelectorAll('.qa-item');
items.forEach(item => {
const isCorrect = item.dataset.correct === 'true';
if (filter === 'all') {
item.style.display = 'block';
} else if (filter === 'correct' && isCorrect) {
item.style.display = 'block';
} else if (filter === 'incorrect' && !isCorrect) {
item.style.display = 'block';
} else {
item.style.display = 'none';
}
});
}

View file

@ -0,0 +1,46 @@
"""
Example: Running many think operations in parallel with optimized connection pooling.
For 100 parallel think operations:
- Each think does 3 searches (world, agent, opinion)
- Each search acquires 1-3 connections briefly
- Total: ~300 concurrent connection requests
Solution: Increase pool_max_size to handle the concurrency.
"""
import asyncio
from memora import TemporalSemanticMemory
async def main():
# For 100 parallel think operations, use a larger pool
# Rule of thumb: pool_max_size >= (num_parallel_thinks * 3)
memory = TemporalSemanticMemory(
pool_min_size=10, # Keep some connections warm
pool_max_size=200 # Allow up to 200 concurrent connections
)
await memory.initialize()
# Example: Run 100 think operations in parallel
queries = [f"Query {i}" for i in range(100)]
tasks = [
memory.think_async(
agent_id="test_agent",
query=query,
thinking_budget=50,
top_k=10
)
for query in queries
]
# Run all thinks in parallel
results = await asyncio.gather(*tasks)
print(f"Completed {len(results)} think operations")
await memory.close()
if __name__ == "__main__":
asyncio.run(main())

166
scripts/profile_queries.py Normal file
View file

@ -0,0 +1,166 @@
"""
Profile slow database queries to identify optimization opportunities.
Usage:
uv run python scripts/profile_queries.py
"""
import asyncio
import os
import asyncpg
from dotenv import load_dotenv
load_dotenv()
async def profile_entry_points_query():
"""Profile the vector similarity entry points query."""
db_url = os.getenv("DATABASE_URL")
conn = await asyncpg.connect(db_url)
# Generate a dummy embedding vector (384 dimensions for bge-small-en-v1.5)
dummy_embedding = str([0.1] * 384)
print("=" * 80)
print("PROFILING: Entry Points Query (Vector Similarity)")
print("=" * 80)
# Run EXPLAIN ANALYZE
explain = await conn.fetch("""
EXPLAIN (ANALYZE, BUFFERS, VERBOSE)
SELECT id, text, context, event_date, access_count, embedding,
1 - (embedding <=> $1::vector) AS similarity
FROM memory_units
WHERE agent_id = $2
AND embedding IS NOT NULL
AND (1 - (embedding <=> $1::vector)) >= 0.5
ORDER BY embedding <=> $1::vector
LIMIT 3
""", dummy_embedding, "test_agent")
for row in explain:
print(row[0])
await conn.close()
async def profile_neighbors_query(sample_node_ids):
"""Profile the neighbors JOIN query."""
db_url = os.getenv("DATABASE_URL")
conn = await asyncpg.connect(db_url)
print("\n" + "=" * 80)
print("PROFILING: Neighbors Query (Graph Traversal)")
print(f"Sample size: {len(sample_node_ids)} nodes")
print("=" * 80)
# Run EXPLAIN ANALYZE
explain = await conn.fetch("""
EXPLAIN (ANALYZE, BUFFERS, VERBOSE)
SELECT ml.from_unit_id, ml.to_unit_id, ml.weight, ml.link_type, ml.entity_id,
mu.text, mu.context, mu.event_date, mu.access_count,
mu.id as neighbor_id
FROM memory_links ml
JOIN memory_units mu ON ml.to_unit_id = mu.id
WHERE ml.from_unit_id = ANY($1::uuid[])
AND ml.weight >= 0.1
ORDER BY ml.from_unit_id, ml.weight DESC
""", sample_node_ids)
for row in explain:
print(row[0])
await conn.close()
async def profile_embeddings_query(sample_node_ids):
"""Profile the batch embeddings fetch query."""
db_url = os.getenv("DATABASE_URL")
conn = await asyncpg.connect(db_url)
print("\n" + "=" * 80)
print("PROFILING: Embeddings Query (Batch Fetch)")
print(f"Sample size: {len(sample_node_ids)} nodes")
print("=" * 80)
# Run EXPLAIN ANALYZE
explain = await conn.fetch("""
EXPLAIN (ANALYZE, BUFFERS, VERBOSE)
SELECT id, embedding
FROM memory_units
WHERE id = ANY($1::uuid[])
""", sample_node_ids)
for row in explain:
print(row[0])
await conn.close()
async def get_sample_node_ids(batch_size=50):
"""Get sample node IDs for profiling."""
db_url = os.getenv("DATABASE_URL")
conn = await asyncpg.connect(db_url)
rows = await conn.fetch(f"""
SELECT id FROM memory_units
LIMIT {batch_size}
""")
await conn.close()
return [row['id'] for row in rows]
async def check_indexes():
"""Check what indexes exist."""
db_url = os.getenv("DATABASE_URL")
conn = await asyncpg.connect(db_url)
print("\n" + "=" * 80)
print("CURRENT INDEXES")
print("=" * 80)
indexes = await conn.fetch("""
SELECT
tablename,
indexname,
indexdef
FROM pg_indexes
WHERE schemaname = 'public'
AND tablename IN ('memory_units', 'memory_links')
ORDER BY tablename, indexname
""")
for idx in indexes:
print(f"\nTable: {idx['tablename']}")
print(f"Index: {idx['indexname']}")
print(f"Definition: {idx['indexdef']}")
await conn.close()
async def main():
print("Starting Query Profiling...")
# Check indexes first
await check_indexes()
# Get sample node IDs
sample_ids = await get_sample_node_ids(50)
if sample_ids:
print(f"\nGot {len(sample_ids)} sample node IDs for profiling")
# Profile each query type
await profile_entry_points_query()
await profile_neighbors_query(sample_ids)
await profile_embeddings_query(sample_ids)
else:
print("\nNo data in database - run ingestion first")
print("\n" + "=" * 80)
print("PROFILING COMPLETE")
print("=" * 80)
if __name__ == "__main__":
asyncio.run(main())