perf think improvements
This commit is contained in:
parent
ba048ff96c
commit
d8fdbb94ef
10 changed files with 1227 additions and 0 deletions
356
benchmarks/locomo/benchmark_results_think.json
Normal file
356
benchmarks/locomo/benchmark_results_think.json
Normal file
|
|
@ -0,0 +1,356 @@
|
|||
{
|
||||
"overall_accuracy": 100.0,
|
||||
"total_correct": 1,
|
||||
"total_questions": 1,
|
||||
"num_items": 1,
|
||||
"item_results": [
|
||||
{
|
||||
"item_id": "conv-26",
|
||||
"metrics": {
|
||||
"accuracy": 100.0,
|
||||
"correct": 1,
|
||||
"total": 1,
|
||||
"category_stats": {
|
||||
"2": {
|
||||
"correct": 1,
|
||||
"total": 1
|
||||
}
|
||||
},
|
||||
"detailed_results": [
|
||||
{
|
||||
"question": "When did Caroline go to the LGBTQ support group?",
|
||||
"correct_answer": "7 May 2023",
|
||||
"predicted_answer": "Caroline attended the LGBTQ support group on May\u202f7\u202f2023 (event recorded at 2023\u201105\u201107T13:56:00\u202fUTC).",
|
||||
"reasoning": "Think API: 20 world facts, 0 agent facts, 20 opinions",
|
||||
"category": 2,
|
||||
"retrieved_memories": [
|
||||
{
|
||||
"id": "5ddabf72-967c-4bd0-a0a8-1545324e48f2",
|
||||
"text": "Caroline attended an LGBTQ support group on 2023-05-07 and found it powerful.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_1)",
|
||||
"event_date": "2023-05-07T13:56:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "7b85f829-743e-4827-8349-ca6d01dba5dc",
|
||||
"text": "Caroline volunteered at an LGBTQ+ youth center",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_15)",
|
||||
"event_date": "2023-08-28T15:19:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "6794d0e8-5e4d-44bb-8238-21249021d101",
|
||||
"text": "Caroline joined the LGBTQ activist group \"Connected LGBTQ Activists\" on 2023-07-18.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_10)",
|
||||
"event_date": "2023-07-18T20:56:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "eb8952a1-9b7d-45b3-8f63-2411d6421725",
|
||||
"text": "Caroline tried to apologize to the people she encountered during the hike.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_14)",
|
||||
"event_date": "2023-08-18T13:33:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "2decc740-9674-4b25-acf1-1da8ba8f61dd",
|
||||
"text": "Caroline owns a guinea pig named Oscar.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_13)",
|
||||
"event_date": "2023-08-23T15:31:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "1aad5731-397b-4c3f-b78d-58768d376133",
|
||||
"text": "Caroline told Melanie that she is lucky to have such an awesome family.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_10)",
|
||||
"event_date": "2023-07-20T20:56:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "5b3e87cc-6892-4fb2-945e-5921a32c3453",
|
||||
"text": "Caroline attended a poetry reading on Friday, October 6, 2023.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_17)",
|
||||
"event_date": "2023-10-06T10:31:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "3fd7f709-6b78-443c-8768-abeee8f28e31",
|
||||
"text": "Caroline created a self-portrait last week and posted a photo of it.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_13)",
|
||||
"event_date": "2023-08-16T15:31:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "640e4293-755b-4634-b570-1f15128eb1e6",
|
||||
"text": "The event room was electric with energy and support, and the posters displayed pride and strength, which inspired Caroline to create new artwork.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_17)",
|
||||
"event_date": "2023-10-06T10:31:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "a9331948-bf5f-4791-86e9-0bf62342ed88",
|
||||
"text": "Caroline visited the beach.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_14)",
|
||||
"event_date": "2023-08-18T13:33:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "c48b29f0-bcb4-4fde-b616-4e08d9166146",
|
||||
"text": "Caroline attended an adoption advice and assistance group, receiving a lot of help.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_13)",
|
||||
"event_date": "2023-08-23T15:31:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "eba11741-e2ae-4ccf-976a-4df864b0cace",
|
||||
"text": "Caroline heard transgender stories at the LGBTQ support group, which she found inspiring.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_1)",
|
||||
"event_date": "2023-05-07T13:56:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "5968eee3-380d-4780-afec-9576fbafe422",
|
||||
"text": "Caroline is interested in a career in counseling or mental health to support people with similar issues.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_1)",
|
||||
"event_date": "2023-05-08T13:56:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "aee0214f-e49d-47a4-83f5-5f97304a3a51",
|
||||
"text": "Caroline feels supported by people around her, which makes her feel okay.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_12)",
|
||||
"event_date": "2023-08-17T13:50:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "149347b4-b7cb-4bf9-a8be-f743700a9783",
|
||||
"text": "The city held a pride parade on 2023-07-15, where many people marched, waved flags, held signs, and celebrated love and diversity; Caroline missed the parade.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_10)",
|
||||
"event_date": "2023-07-15T20:56:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "21ac438b-ce0b-4efd-8c6e-d424295ec3cc",
|
||||
"text": "Caroline created a recent painting that represents inclusivity and diversity and uses it to speak up for the LGBTQ+ community and push for acceptance.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_11)",
|
||||
"event_date": "2023-08-14T14:24:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "442b4482-f31b-4ef5-b545-eb0befab7cfe",
|
||||
"text": "Caroline mentioned an advocacy event that was a cool experience with love and support.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_11)",
|
||||
"event_date": "2023-08-14T14:24:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "a100f54b-7112-4616-8f7c-9a7347587597",
|
||||
"text": "Caroline had a not-so-great experience on a hike where she ran into a group of religious conservatives who said something that upset her, leading her to reflect on the need for more work on LGBTQ rights.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_12)",
|
||||
"event_date": "2023-08-17T13:50:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "3e2e6418-29db-44e4-963f-ea2089d33dd5",
|
||||
"text": "Caroline has a lifelong love for horses.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_13)",
|
||||
"event_date": "2023-08-23T15:31:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "0dda577c-9ea3-4a47-835a-63cade830838",
|
||||
"text": "Caroline gave a talk at her school event last week about her transgender journey, encouraging students to get involved in the LGBTQ community and observed positive reactions from the audience.",
|
||||
"context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_3)",
|
||||
"event_date": "2023-06-02T19:55:00+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "world"
|
||||
},
|
||||
{
|
||||
"id": "054d8d37-aa7d-4a21-b4bd-7968e29d74dd",
|
||||
"text": "Caroline is a participant within the LGBTQ community",
|
||||
"context": "formed during thinking about: In what ways is Caroline participating in the LGBTQ community?",
|
||||
"event_date": "2025-11-04T18:24:51+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "9862a174-0e48-418c-b796-2cb1185b7f61",
|
||||
"text": "Caroline's plan to build a personal library for her future children suggests she may consider formal study in library science or related fields.",
|
||||
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
|
||||
"event_date": "2025-11-04T15:30:15+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "7318bff4-0702-44b9-8aa6-b50f5a59be86",
|
||||
"text": "We cannot determine whether Caroline wants to move back to her home country soon",
|
||||
"context": "formed during thinking about: Would Caroline want to move back to her home country soon?",
|
||||
"event_date": "2025-11-04T18:05:51+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "cdb11bda-7c70-4b61-b36e-bdf21727a194",
|
||||
"text": "Caroline is an event organizer within the LGBTQ community",
|
||||
"context": "formed during thinking about: In what ways is Caroline participating in the LGBTQ community?",
|
||||
"event_date": "2025-11-04T18:24:51+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "1d68207b-03e2-44ca-a79b-47ba427c1ba1",
|
||||
"text": "Melanie and Caroline describe their journey as collaborative, supportive, and inspiring",
|
||||
"context": "formed during thinking about: How do Melanie and Caroline describe their journey through life together?",
|
||||
"event_date": "2025-11-04T18:24:08+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "8b6a13ff-7417-4458-a19c-cd101d24c53f",
|
||||
"text": "Caroline has passed adoption agency interviews.",
|
||||
"context": "formed during thinking about: What is Caroline's identity?",
|
||||
"event_date": "2025-11-04T15:30:19+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "4e8a04ef-ea96-48ec-b016-c32a111bc819",
|
||||
"text": "Caroline is an advocate within the LGBTQ community",
|
||||
"context": "formed during thinking about: In what ways is Caroline participating in the LGBTQ community?",
|
||||
"event_date": "2025-11-04T18:24:51+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "87c97473-6596-4e5c-bd0f-f989ab76e137",
|
||||
"text": "The provided facts only describe Caroline's activities, goals, and interests in her current location",
|
||||
"context": "formed during thinking about: Would Caroline want to move back to her home country soon?",
|
||||
"event_date": "2025-11-04T18:05:51+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "6aa8c0fd-dffc-4d27-9559-b0687d263f59",
|
||||
"text": "Caroline plays piano.",
|
||||
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
|
||||
"event_date": "2025-11-04T15:27:06+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "04372c16-57ae-429b-8f9c-e3ab9ef3bacd",
|
||||
"text": "Caroline could integrate art therapy classes into counseling work.",
|
||||
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
|
||||
"event_date": "2025-11-04T15:27:06+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "d596524c-fdd6-4080-b7a7-f9da2cd3d6ed",
|
||||
"text": "Caroline explicitly said she wants to continue her education.",
|
||||
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
|
||||
"event_date": "2025-11-04T18:08:28+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "f5308bcf-a59e-4b1d-90e0-0e243b79b98d",
|
||||
"text": "Caroline paints.",
|
||||
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
|
||||
"event_date": "2025-11-04T15:27:06+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "f97a5f20-1e8d-45bc-93db-6626070ca5f1",
|
||||
"text": "Caroline is interested in a career in mental-health services.",
|
||||
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
|
||||
"event_date": "2025-11-04T18:08:28+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "fc805a93-3b1d-4f8a-a98e-3b7044bcb2d0",
|
||||
"text": "Caroline feels a sense of empowerment from activism.",
|
||||
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
|
||||
"event_date": "2025-11-04T15:27:06+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "31c26b9a-f4a3-4a08-bae8-410780b40414",
|
||||
"text": "Previously I thought Caroline would likely focus on LGBTQ+ studies, gender studies, or social justice based on her activism. However, her explicit statements about pursuing a career in counseling and mental health, and her desire to support people with similar issues, indicate that her primary focus is now likely on a counseling or mental\u2011health career, potentially integrating her advocacy for marginalized communities.",
|
||||
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
|
||||
"event_date": "2025-11-04T15:27:06+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "987325a5-6b67-42bf-8910-fd194987bd35",
|
||||
"text": "Caroline is a future parent-to-be.",
|
||||
"context": "formed during thinking about: What is Caroline's identity?",
|
||||
"event_date": "2025-11-04T15:30:19+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "ba5ae772-2d00-4cbf-ac88-6988bca1be78",
|
||||
"text": "Melanie and Caroline frame challenges as a shared roller-coaster",
|
||||
"context": "formed during thinking about: How do Melanie and Caroline describe their journey through life together?",
|
||||
"event_date": "2025-11-04T18:24:08+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "de3b98e6-f435-41b7-a38b-712a35cd0c54",
|
||||
"text": "There is no information indicating Caroline identifies as religious.",
|
||||
"context": "formed during thinking about: Would Caroline be considered religious?",
|
||||
"event_date": "2025-11-04T18:05:26+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "314d9820-e75d-455f-bfad-55dcea90c044",
|
||||
"text": "Caroline may choose a multidisciplinary program that combines counseling/mental\u2011health training, LGBTQ+ advocacy, and early childhood literacy.",
|
||||
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
|
||||
"event_date": "2025-11-04T15:30:15+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
},
|
||||
{
|
||||
"id": "5ba2cc08-4998-4667-8b95-adb1f0986504",
|
||||
"text": "Caroline is most likely to pursue LGBTQ+ Studies and Gender & Social Justice.",
|
||||
"context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?",
|
||||
"event_date": "2025-11-04T15:30:15+00:00",
|
||||
"score": 0.0,
|
||||
"fact_type": "opinion"
|
||||
}
|
||||
],
|
||||
"is_correct": true,
|
||||
"correctness_reasoning": "The generated answer states the same date (May\u202f7\u202f2023) as the gold answer, so it is correct."
|
||||
}
|
||||
]
|
||||
},
|
||||
"num_sessions": -1
|
||||
}
|
||||
]
|
||||
}
|
||||
7
benchmarks/locomo/results_table_think.md
Normal file
7
benchmarks/locomo/results_table_think.md
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
# LoComo Benchmark Results (Think Mode)
|
||||
|
||||
**Overall Accuracy**: 100.00% (1/1)
|
||||
|
||||
| Sample ID | Sessions | Questions | Correct | Accuracy | Multi-hop | Single-hop | Temporal | Open-domain |
|
||||
|-----------|----------|-----------|---------|----------|-----------|------------|----------|-------------|
|
||||
| conv-26 | -1 | 1 | 1 | 100.00% | N/A | N/A | N/A | N/A |
|
||||
62
benchmarks/visualizer/README.md
Normal file
62
benchmarks/visualizer/README.md
Normal file
|
|
@ -0,0 +1,62 @@
|
|||
# Benchmark Visualizer
|
||||
|
||||
A standalone web service for visualizing benchmark results. Currently supports the LoComo benchmark with plans to add more benchmarks in the future.
|
||||
|
||||
## Features
|
||||
|
||||
- Interactive web interface for viewing benchmark results
|
||||
- Detailed breakdown by category (Multi-hop, Single-hop, Temporal, Open-domain)
|
||||
- Filter options to view all, correct, or incorrect answers
|
||||
- Expandable Q&A details with reasoning and retrieved memories
|
||||
- Overall and per-item accuracy statistics
|
||||
|
||||
## Running the Visualizer
|
||||
|
||||
### Option 1: Using the serve script (recommended)
|
||||
|
||||
```bash
|
||||
cd benchmarks/visualizer
|
||||
./serve.sh
|
||||
```
|
||||
|
||||
### Option 2: Using uvicorn directly
|
||||
|
||||
```bash
|
||||
cd benchmarks/visualizer
|
||||
uv run uvicorn server:app --reload --host 0.0.0.0 --port 8001
|
||||
```
|
||||
|
||||
Then open your browser to: http://localhost:8001
|
||||
|
||||
## Usage
|
||||
|
||||
1. Select a benchmark from the dropdown:
|
||||
- **LoComo (search)**: Traditional two-step approach (search → LLM answer generation)
|
||||
- **LoComo (think)**: Integrated approach using think API (single call for retrieval + reasoning)
|
||||
2. The visualization will automatically load and display:
|
||||
- Overall accuracy statistics
|
||||
- Category-wise performance breakdown
|
||||
- Detailed results for each conversation
|
||||
3. Use the filter controls to show all answers, only incorrect, or only correct answers
|
||||
4. Expand individual conversations to see Q&A details, reasoning, and retrieved memories
|
||||
|
||||
## API Endpoints
|
||||
|
||||
- `GET /` - Main visualizer page
|
||||
- `GET /api/locomo?mode={search|think}` - Returns LoComo benchmark results as JSON
|
||||
- `mode=search` (default): Returns results from `benchmark_results.json`
|
||||
- `mode=think`: Returns results from `benchmark_results_think.json`
|
||||
|
||||
## Requirements
|
||||
|
||||
- FastAPI
|
||||
- Uvicorn
|
||||
- Python 3.11+
|
||||
|
||||
The visualizer reads benchmark results from:
|
||||
- `benchmarks/locomo/benchmark_results.json` for search mode
|
||||
- `benchmarks/locomo/benchmark_results_think.json` for think mode
|
||||
|
||||
Make sure to run the benchmark first to generate results:
|
||||
- Search mode: `cd benchmarks/locomo && uv run python run_benchmark.py`
|
||||
- Think mode: `cd benchmarks/locomo && uv run python run_benchmark.py --use-think`
|
||||
4
benchmarks/visualizer/serve.sh
Executable file
4
benchmarks/visualizer/serve.sh
Executable file
|
|
@ -0,0 +1,4 @@
|
|||
#!/bin/bash
|
||||
# Start the Benchmark Visualizer server with hot reload
|
||||
cd "$(dirname "$0")"
|
||||
uv run uvicorn server:app --reload --host 0.0.0.0 --port 8001
|
||||
72
benchmarks/visualizer/server.py
Normal file
72
benchmarks/visualizer/server.py
Normal file
|
|
@ -0,0 +1,72 @@
|
|||
"""Benchmark Visualizer Web Service.
|
||||
|
||||
A standalone web service for visualizing benchmark results.
|
||||
Currently supports LoComo benchmark visualization.
|
||||
"""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from fastapi import FastAPI, HTTPException
|
||||
from fastapi.responses import HTMLResponse, FileResponse
|
||||
from fastapi.staticfiles import StaticFiles
|
||||
|
||||
app = FastAPI(title="Benchmark Visualizer")
|
||||
|
||||
# Get the benchmarks directory
|
||||
BENCHMARKS_DIR = Path(__file__).parent.parent
|
||||
|
||||
|
||||
@app.get("/", response_class=HTMLResponse)
|
||||
async def index():
|
||||
"""Serve the main benchmark visualizer page."""
|
||||
html_path = Path(__file__).parent / "static" / "index.html"
|
||||
with open(html_path) as f:
|
||||
return f.read()
|
||||
|
||||
|
||||
@app.get("/api/locomo")
|
||||
async def get_locomo_results(mode: str = "search") -> dict[str, Any]:
|
||||
"""Get LoComo benchmark results.
|
||||
|
||||
Returns pre-computed benchmark results from the locomo directory.
|
||||
|
||||
Args:
|
||||
mode: Either "search" (default) or "think" to select which results to load
|
||||
"""
|
||||
try:
|
||||
# Determine filename based on mode
|
||||
if mode == "think":
|
||||
filename = "benchmark_results_think.json"
|
||||
else:
|
||||
filename = "benchmark_results.json"
|
||||
|
||||
results_path = BENCHMARKS_DIR / "locomo" / filename
|
||||
|
||||
if not results_path.exists():
|
||||
raise HTTPException(
|
||||
status_code=404,
|
||||
detail=f"Benchmark results not found for mode '{mode}'. Please run the benchmark first with {'--use-think' if mode == 'think' else 'default settings'}."
|
||||
)
|
||||
|
||||
with open(results_path) as f:
|
||||
results = json.load(f)
|
||||
|
||||
return results
|
||||
except json.JSONDecodeError as e:
|
||||
raise HTTPException(
|
||||
status_code=500,
|
||||
detail=f"Failed to parse benchmark results: {str(e)}"
|
||||
)
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
|
||||
# Mount static files
|
||||
app.mount("/static", StaticFiles(directory=Path(__file__).parent / "static"), name="static")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import uvicorn
|
||||
uvicorn.run(app, host="127.0.0.1", port=8001)
|
||||
140
benchmarks/visualizer/static/css/styles.css
Normal file
140
benchmarks/visualizer/static/css/styles.css
Normal file
|
|
@ -0,0 +1,140 @@
|
|||
body {
|
||||
font-family: Tahoma, sans-serif;
|
||||
margin: 0;
|
||||
padding: 0;
|
||||
background: #f5f5f5;
|
||||
}
|
||||
|
||||
.header {
|
||||
background: #333;
|
||||
color: white;
|
||||
padding: 20px;
|
||||
border-bottom: 3px solid #42a5f5;
|
||||
}
|
||||
|
||||
.header h1 {
|
||||
margin: 0 0 5px 0;
|
||||
}
|
||||
|
||||
.header p {
|
||||
margin: 0;
|
||||
color: #ccc;
|
||||
}
|
||||
|
||||
.benchmark-selector {
|
||||
background: #f0f0f0;
|
||||
padding: 15px 20px;
|
||||
border-bottom: 2px solid #333;
|
||||
display: flex;
|
||||
gap: 15px;
|
||||
align-items: center;
|
||||
}
|
||||
|
||||
.benchmark-selector label {
|
||||
font-weight: bold;
|
||||
font-size: 14px;
|
||||
}
|
||||
|
||||
.benchmark-selector select {
|
||||
padding: 8px 12px;
|
||||
border: 2px solid #42a5f5;
|
||||
border-radius: 4px;
|
||||
background: white;
|
||||
color: #333;
|
||||
font-size: 14px;
|
||||
font-weight: bold;
|
||||
cursor: pointer;
|
||||
min-width: 200px;
|
||||
}
|
||||
|
||||
#benchmark-content {
|
||||
padding: 20px;
|
||||
}
|
||||
|
||||
.welcome-message {
|
||||
text-align: center;
|
||||
padding: 60px 20px;
|
||||
color: #666;
|
||||
}
|
||||
|
||||
.welcome-message h2 {
|
||||
color: #333;
|
||||
}
|
||||
|
||||
.load-button {
|
||||
padding: 8px 20px;
|
||||
background: #66bb6a;
|
||||
color: white;
|
||||
border: none;
|
||||
border-radius: 4px;
|
||||
cursor: pointer;
|
||||
font-weight: bold;
|
||||
font-size: 14px;
|
||||
margin-bottom: 20px;
|
||||
}
|
||||
|
||||
.load-button:hover {
|
||||
background: #43a047;
|
||||
}
|
||||
|
||||
.error-message {
|
||||
color: #d32f2f;
|
||||
padding: 20px;
|
||||
background: #ffebee;
|
||||
border: 2px solid #ef5350;
|
||||
border-radius: 8px;
|
||||
margin: 20px;
|
||||
max-width: 800px;
|
||||
}
|
||||
|
||||
.error-message h3 {
|
||||
margin-top: 0;
|
||||
color: #c62828;
|
||||
}
|
||||
|
||||
.error-message pre {
|
||||
background: #f5f5f5;
|
||||
padding: 10px;
|
||||
border-radius: 4px;
|
||||
overflow-x: auto;
|
||||
color: #333;
|
||||
font-family: monospace;
|
||||
font-size: 13px;
|
||||
}
|
||||
|
||||
.stats-grid {
|
||||
display: grid;
|
||||
grid-template-columns: repeat(auto-fit, minmax(200px, 1fr));
|
||||
gap: 10px;
|
||||
margin-top: 10px;
|
||||
}
|
||||
|
||||
.stat-item {
|
||||
padding: 12px;
|
||||
background: white;
|
||||
border: 1px solid #ddd;
|
||||
border-radius: 4px;
|
||||
text-align: center;
|
||||
}
|
||||
|
||||
.stat-label {
|
||||
font-weight: bold;
|
||||
color: #666;
|
||||
font-size: 12px;
|
||||
margin-bottom: 5px;
|
||||
}
|
||||
|
||||
.stat-value {
|
||||
font-size: 24px;
|
||||
color: #333;
|
||||
font-weight: bold;
|
||||
}
|
||||
|
||||
.qa-results {
|
||||
margin-top: 20px;
|
||||
}
|
||||
|
||||
.qa-item {
|
||||
margin-bottom: 15px;
|
||||
border-radius: 8px;
|
||||
}
|
||||
32
benchmarks/visualizer/static/index.html
Normal file
32
benchmarks/visualizer/static/index.html
Normal file
|
|
@ -0,0 +1,32 @@
|
|||
<!DOCTYPE html>
|
||||
<html>
|
||||
<head>
|
||||
<title>Benchmark Visualizer</title>
|
||||
<meta charset="utf-8">
|
||||
<link rel="stylesheet" href="/static/css/styles.css">
|
||||
</head>
|
||||
<body>
|
||||
<div class="header">
|
||||
<h1>Benchmark Visualizer</h1>
|
||||
<p>Analyze and visualize benchmark results</p>
|
||||
</div>
|
||||
|
||||
<div class="benchmark-selector">
|
||||
<label>Select Benchmark:</label>
|
||||
<select id="benchmark-select" onchange="selectBenchmark()">
|
||||
<option value="">-- Select a benchmark --</option>
|
||||
<option value="locomo-search">LoComo (search)</option>
|
||||
<option value="locomo-think">LoComo (think)</option>
|
||||
</select>
|
||||
</div>
|
||||
|
||||
<div id="benchmark-content">
|
||||
<div class="welcome-message">
|
||||
<h2>Welcome to Benchmark Visualizer</h2>
|
||||
<p>Select a benchmark from the dropdown above to view results.</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<script src="/static/js/app.js"></script>
|
||||
</body>
|
||||
</html>
|
||||
342
benchmarks/visualizer/static/js/app.js
Normal file
342
benchmarks/visualizer/static/js/app.js
Normal file
|
|
@ -0,0 +1,342 @@
|
|||
// Benchmark Visualizer App
|
||||
|
||||
let currentBenchmark = null;
|
||||
let benchmarkData = null;
|
||||
|
||||
function selectBenchmark() {
|
||||
const select = document.getElementById('benchmark-select');
|
||||
currentBenchmark = select.value;
|
||||
|
||||
if (!currentBenchmark) {
|
||||
document.getElementById('benchmark-content').innerHTML = `
|
||||
<div class="welcome-message">
|
||||
<h2>Welcome to Benchmark Visualizer</h2>
|
||||
<p>Select a benchmark from the dropdown above to view results.</p>
|
||||
</div>
|
||||
`;
|
||||
return;
|
||||
}
|
||||
|
||||
// Load the selected benchmark
|
||||
if (currentBenchmark === 'locomo-search') {
|
||||
loadLocomoResults('search');
|
||||
} else if (currentBenchmark === 'locomo-think') {
|
||||
loadLocomoResults('think');
|
||||
}
|
||||
}
|
||||
|
||||
async function loadLocomoResults(mode = 'search') {
|
||||
try {
|
||||
const response = await fetch(`/api/locomo?mode=${mode}`);
|
||||
|
||||
if (!response.ok) {
|
||||
const errorData = await response.json();
|
||||
const modeLabel = mode === 'think' ? 'think' : 'search';
|
||||
const runCommand = mode === 'think'
|
||||
? 'uv run python run_benchmark.py --use-think'
|
||||
: 'uv run python run_benchmark.py';
|
||||
|
||||
document.getElementById('benchmark-content').innerHTML = `
|
||||
<div class="error-message">
|
||||
<h3>⚠️ Benchmark Results Not Found</h3>
|
||||
<p>${errorData.detail || 'The requested benchmark results are not available.'}</p>
|
||||
<p><strong>To generate ${modeLabel} mode results:</strong></p>
|
||||
<pre style="background: #f5f5f5; padding: 10px; border-radius: 4px; overflow-x: auto;">cd benchmarks/locomo
|
||||
${runCommand}</pre>
|
||||
<p style="margin-top: 15px; font-size: 14px; color: #666;">
|
||||
Once the benchmark completes, refresh this page and select "${mode === 'think' ? 'LoComo (think)' : 'LoComo (search)'}" again.
|
||||
</p>
|
||||
</div>
|
||||
`;
|
||||
return;
|
||||
}
|
||||
|
||||
benchmarkData = await response.json();
|
||||
console.log(`Loaded locomo data (${mode} mode):`, benchmarkData);
|
||||
renderLocomoResults(mode);
|
||||
} catch (e) {
|
||||
console.error('Error loading benchmark results:', e);
|
||||
document.getElementById('benchmark-content').innerHTML = `
|
||||
<div class="error-message">
|
||||
<h3>❌ Error Loading Results</h3>
|
||||
<p>${e.message}</p>
|
||||
<p style="font-size: 12px; color: #666; margin-top: 10px;">Check the browser console for more details.</p>
|
||||
</div>
|
||||
`;
|
||||
}
|
||||
}
|
||||
|
||||
function renderLocomoResults(mode = 'search') {
|
||||
if (!benchmarkData) return;
|
||||
|
||||
const content = document.getElementById('benchmark-content');
|
||||
|
||||
try {
|
||||
// Handle both old and new structure
|
||||
const results = benchmarkData.item_results || benchmarkData.conversation_results || [];
|
||||
const numItems = benchmarkData.num_items || results.length;
|
||||
|
||||
console.log('Rendering results:', { resultsCount: results.length, numItems });
|
||||
|
||||
// Calculate per-category statistics
|
||||
const categoryStats = {
|
||||
1: { name: 'Multi-hop', correct: 0, total: 0 },
|
||||
2: { name: 'Single-hop', correct: 0, total: 0 },
|
||||
3: { name: 'Temporal', correct: 0, total: 0 },
|
||||
4: { name: 'Open-domain', correct: 0, total: 0 }
|
||||
};
|
||||
|
||||
// Aggregate across all items
|
||||
results.forEach(item => {
|
||||
if (item.metrics && item.metrics.detailed_results) {
|
||||
item.metrics.detailed_results.forEach(result => {
|
||||
const category = result.category;
|
||||
if (categoryStats[category]) {
|
||||
categoryStats[category].total++;
|
||||
if (result.is_correct) {
|
||||
categoryStats[category].correct++;
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
// Determine title based on mode
|
||||
const modeLabel = mode === 'think' ? ' (Think Mode)' : ' (Search Mode)';
|
||||
|
||||
// Overall stats
|
||||
const overallHtml = `
|
||||
<div style="background: #f9f9f9; padding: 20px; border: 2px solid #333; border-radius: 8px; margin-bottom: 20px;">
|
||||
<h3 style="margin-top: 0;">LoComo Benchmark${modeLabel} - Overall Performance</h3>
|
||||
<div class="stats-grid">
|
||||
<div class="stat-item">
|
||||
<div class="stat-label">Overall Accuracy</div>
|
||||
<div class="stat-value">${benchmarkData.overall_accuracy.toFixed(2)}%</div>
|
||||
</div>
|
||||
<div class="stat-item">
|
||||
<div class="stat-label">Correct Answers</div>
|
||||
<div class="stat-value">${benchmarkData.total_correct} / ${benchmarkData.total_questions}</div>
|
||||
</div>
|
||||
<div class="stat-item">
|
||||
<div class="stat-label">Items</div>
|
||||
<div class="stat-value">${numItems}</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<h4 style="margin: 20px 0 10px 0; padding-top: 15px; border-top: 1px solid #ddd;">Accuracy by Category</h4>
|
||||
<div class="stats-grid" style="grid-template-columns: repeat(auto-fit, minmax(200px, 1fr));">
|
||||
${Object.values(categoryStats).map(cat => {
|
||||
const accuracy = cat.total > 0 ? ((cat.correct / cat.total) * 100).toFixed(1) : 0;
|
||||
const color = accuracy >= 70 ? '#43a047' : accuracy >= 50 ? '#ff9800' : '#e53935';
|
||||
return `
|
||||
<div class="stat-item">
|
||||
<div class="stat-label">${cat.name}</div>
|
||||
<div class="stat-value" style="color: ${color};">${accuracy}%</div>
|
||||
<div style="font-size: 11px; color: #666; margin-top: 4px;">${cat.correct} / ${cat.total}</div>
|
||||
</div>
|
||||
`;
|
||||
}).join('')}
|
||||
</div>
|
||||
</div>
|
||||
`;
|
||||
|
||||
// Filter controls
|
||||
const filterHtml = `
|
||||
<div style="margin-bottom: 20px; display: flex; gap: 10px; align-items: center;">
|
||||
<label style="font-weight: bold;">Show:</label>
|
||||
<label><input type="radio" name="answer-filter" value="all" checked onchange="filterAnswers()"> All Answers</label>
|
||||
<label><input type="radio" name="answer-filter" value="incorrect" onchange="filterAnswers()"> ❌ Incorrect Only</label>
|
||||
<label><input type="radio" name="answer-filter" value="correct" onchange="filterAnswers()"> ✅ Correct Only</label>
|
||||
</div>
|
||||
`;
|
||||
|
||||
// Build item sections
|
||||
let itemsHtml = '';
|
||||
results.forEach((item, idx) => {
|
||||
const itemId = item.item_id || item.sample_id || `item-${idx}`;
|
||||
const accuracy = item.metrics.accuracy.toFixed(2);
|
||||
const correctCount = item.metrics.correct;
|
||||
const totalCount = item.metrics.total;
|
||||
|
||||
itemsHtml += `
|
||||
<div style="margin-bottom: 30px; border: 2px solid #333; border-radius: 8px; overflow: hidden;">
|
||||
<div style="background: #f0f0f0; padding: 15px; border-bottom: 2px solid #333; cursor: pointer;" onclick="toggleConversation(${idx})">
|
||||
<h3 style="margin: 0; display: flex; justify-content: space-between; align-items: center;">
|
||||
<span>📊 ${itemId}</span>
|
||||
<span style="font-size: 18px; color: ${accuracy >= 70 ? '#43a047' : accuracy >= 50 ? '#ff9800' : '#e53935'};">
|
||||
${accuracy}% (${correctCount}/${totalCount})
|
||||
</span>
|
||||
</h3>
|
||||
</div>
|
||||
<div id="conv-${idx}" style="display: none; padding: 20px;">
|
||||
${renderConversationDetails(item)}
|
||||
</div>
|
||||
</div>
|
||||
`;
|
||||
});
|
||||
|
||||
content.innerHTML = overallHtml + filterHtml + itemsHtml;
|
||||
} catch (e) {
|
||||
console.error('Error rendering Locomo results:', e);
|
||||
content.innerHTML = `
|
||||
<div class="error-message">
|
||||
<strong>Error rendering results:</strong> ${e.message}<br>
|
||||
<pre style="margin-top: 10px; font-size: 11px; overflow: auto;">${e.stack}</pre>
|
||||
</div>
|
||||
`;
|
||||
}
|
||||
}
|
||||
|
||||
function renderConversationDetails(conv) {
|
||||
if (!conv || !conv.metrics) {
|
||||
return '<div style="padding: 20px; color: #666;">No metrics available</div>';
|
||||
}
|
||||
|
||||
const results = conv.metrics.detailed_results;
|
||||
if (!results || !Array.isArray(results) || results.length === 0) {
|
||||
return '<div style="padding: 20px; color: #666;">No detailed results available</div>';
|
||||
}
|
||||
|
||||
let html = '<div class="qa-results">';
|
||||
|
||||
results.forEach((result, idx) => {
|
||||
const isCorrect = result.is_correct;
|
||||
const bgColor = isCorrect ? '#e8f5e9' : '#ffebee';
|
||||
const icon = isCorrect ? '✅' : '❌';
|
||||
const category = getCategoryName(result.category);
|
||||
|
||||
html += `
|
||||
<div class="qa-item" data-correct="${isCorrect}" style="background: ${bgColor}; padding: 15px; margin-bottom: 15px; border: 1px solid #ddd; border-radius: 8px;">
|
||||
<div style="display: flex; justify-content: space-between; align-items: flex-start; margin-bottom: 10px;">
|
||||
<div style="flex: 1;">
|
||||
<div style="font-weight: bold; font-size: 16px; margin-bottom: 8px;">
|
||||
${icon} Question ${idx + 1} <span style="font-size: 12px; background: #666; color: white; padding: 2px 8px; border-radius: 4px; margin-left: 8px;">${category}</span>
|
||||
</div>
|
||||
<div style="margin-bottom: 8px;">
|
||||
<b>Q:</b> ${result.question}
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div style="display: grid; grid-template-columns: 1fr 1fr; gap: 15px; margin-bottom: 10px;">
|
||||
<div>
|
||||
<div style="font-weight: bold; color: #43a047; margin-bottom: 4px;">✓ Correct Answer:</div>
|
||||
<div style="background: white; padding: 8px; border-radius: 4px; border: 1px solid #ccc;">
|
||||
${result.correct_answer}
|
||||
</div>
|
||||
</div>
|
||||
<div>
|
||||
<div style="font-weight: bold; color: ${isCorrect ? '#43a047' : '#e53935'}; margin-bottom: 4px;">
|
||||
${isCorrect ? '✓' : '✗'} Predicted Answer:
|
||||
</div>
|
||||
<div style="background: white; padding: 8px; border-radius: 4px; border: 1px solid #ccc;">
|
||||
${result.predicted_answer}
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<details style="margin-top: 10px;">
|
||||
<summary style="cursor: pointer; font-weight: bold; padding: 5px; background: rgba(255,255,255,0.5); border-radius: 4px;">
|
||||
📝 Show Reasoning & Retrieved Memories
|
||||
</summary>
|
||||
<div style="margin-top: 10px; padding: 10px; background: white; border-radius: 4px;">
|
||||
<div style="margin-bottom: 10px;">
|
||||
<b>System Reasoning:</b>
|
||||
<div style="padding: 8px; background: #f5f5f5; border-radius: 4px; margin-top: 4px;">
|
||||
${result.reasoning}
|
||||
</div>
|
||||
</div>
|
||||
<div style="margin-bottom: 10px;">
|
||||
<b>Judge Reasoning:</b>
|
||||
<div style="padding: 8px; background: #f5f5f5; border-radius: 4px; margin-top: 4px;">
|
||||
${result.correctness_reasoning || 'N/A'}
|
||||
</div>
|
||||
</div>
|
||||
<div>
|
||||
<b>Retrieved Memories (${result.retrieved_memories ? result.retrieved_memories.length : 0}):</b>
|
||||
${renderRetrievedMemories(result.retrieved_memories)}
|
||||
</div>
|
||||
</div>
|
||||
</details>
|
||||
</div>
|
||||
`;
|
||||
});
|
||||
|
||||
html += '</div>';
|
||||
return html;
|
||||
}
|
||||
|
||||
function renderRetrievedMemories(memories) {
|
||||
if (!memories || !Array.isArray(memories) || memories.length === 0) {
|
||||
return '<div style="padding: 8px; color: #999;">No memories retrieved</div>';
|
||||
}
|
||||
|
||||
let html = '<div style="margin-top: 8px;">';
|
||||
memories.forEach((mem, idx) => {
|
||||
if (!mem) return;
|
||||
const eventDate = mem.event_date ? new Date(mem.event_date).toLocaleString() : 'N/A';
|
||||
|
||||
// Determine border color based on fact type
|
||||
let borderColor = '#42a5f5'; // default blue
|
||||
let factTypeLabel = '';
|
||||
if (mem.fact_type) {
|
||||
factTypeLabel = `<span style="background: #666; color: white; padding: 2px 6px; border-radius: 3px; font-size: 10px; margin-left: 8px;">${mem.fact_type.toUpperCase()}</span>`;
|
||||
if (mem.fact_type === 'world') {
|
||||
borderColor = '#4caf50'; // green
|
||||
} else if (mem.fact_type === 'agent') {
|
||||
borderColor = '#ff9800'; // orange
|
||||
} else if (mem.fact_type === 'opinion') {
|
||||
borderColor = '#9c27b0'; // purple
|
||||
}
|
||||
}
|
||||
|
||||
html += `
|
||||
<div style="padding: 8px; background: #f5f5f5; border-left: 3px solid ${borderColor}; margin-bottom: 8px;">
|
||||
<div style="font-size: 11px; color: #666; margin-bottom: 4px;">
|
||||
Rank #${idx + 1} | Score: ${mem.score ? mem.score.toFixed(4) : 'N/A'} | Event Date: ${eventDate}${factTypeLabel}
|
||||
</div>
|
||||
<div style="font-size: 13px;">${mem.text}</div>
|
||||
</div>
|
||||
`;
|
||||
});
|
||||
html += '</div>';
|
||||
return html;
|
||||
}
|
||||
|
||||
function getCategoryName(category) {
|
||||
const categories = {
|
||||
1: 'Multi-hop',
|
||||
2: 'Single-hop',
|
||||
3: 'Temporal',
|
||||
4: 'Open-domain'
|
||||
};
|
||||
return categories[category] || 'Unknown';
|
||||
}
|
||||
|
||||
function toggleConversation(idx) {
|
||||
const elem = document.getElementById(`conv-${idx}`);
|
||||
if (elem.style.display === 'none') {
|
||||
elem.style.display = 'block';
|
||||
} else {
|
||||
elem.style.display = 'none';
|
||||
}
|
||||
}
|
||||
|
||||
function filterAnswers() {
|
||||
const filter = document.querySelector('input[name="answer-filter"]:checked').value;
|
||||
const items = document.querySelectorAll('.qa-item');
|
||||
|
||||
items.forEach(item => {
|
||||
const isCorrect = item.dataset.correct === 'true';
|
||||
|
||||
if (filter === 'all') {
|
||||
item.style.display = 'block';
|
||||
} else if (filter === 'correct' && isCorrect) {
|
||||
item.style.display = 'block';
|
||||
} else if (filter === 'incorrect' && !isCorrect) {
|
||||
item.style.display = 'block';
|
||||
} else {
|
||||
item.style.display = 'none';
|
||||
}
|
||||
});
|
||||
}
|
||||
46
examples/parallel_think.py
Normal file
46
examples/parallel_think.py
Normal file
|
|
@ -0,0 +1,46 @@
|
|||
"""
|
||||
Example: Running many think operations in parallel with optimized connection pooling.
|
||||
|
||||
For 100 parallel think operations:
|
||||
- Each think does 3 searches (world, agent, opinion)
|
||||
- Each search acquires 1-3 connections briefly
|
||||
- Total: ~300 concurrent connection requests
|
||||
|
||||
Solution: Increase pool_max_size to handle the concurrency.
|
||||
"""
|
||||
import asyncio
|
||||
from memora import TemporalSemanticMemory
|
||||
|
||||
|
||||
async def main():
|
||||
# For 100 parallel think operations, use a larger pool
|
||||
# Rule of thumb: pool_max_size >= (num_parallel_thinks * 3)
|
||||
memory = TemporalSemanticMemory(
|
||||
pool_min_size=10, # Keep some connections warm
|
||||
pool_max_size=200 # Allow up to 200 concurrent connections
|
||||
)
|
||||
await memory.initialize()
|
||||
|
||||
# Example: Run 100 think operations in parallel
|
||||
queries = [f"Query {i}" for i in range(100)]
|
||||
|
||||
tasks = [
|
||||
memory.think_async(
|
||||
agent_id="test_agent",
|
||||
query=query,
|
||||
thinking_budget=50,
|
||||
top_k=10
|
||||
)
|
||||
for query in queries
|
||||
]
|
||||
|
||||
# Run all thinks in parallel
|
||||
results = await asyncio.gather(*tasks)
|
||||
|
||||
print(f"Completed {len(results)} think operations")
|
||||
|
||||
await memory.close()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
166
scripts/profile_queries.py
Normal file
166
scripts/profile_queries.py
Normal file
|
|
@ -0,0 +1,166 @@
|
|||
"""
|
||||
Profile slow database queries to identify optimization opportunities.
|
||||
|
||||
Usage:
|
||||
uv run python scripts/profile_queries.py
|
||||
"""
|
||||
import asyncio
|
||||
import os
|
||||
import asyncpg
|
||||
from dotenv import load_dotenv
|
||||
|
||||
load_dotenv()
|
||||
|
||||
|
||||
async def profile_entry_points_query():
|
||||
"""Profile the vector similarity entry points query."""
|
||||
db_url = os.getenv("DATABASE_URL")
|
||||
conn = await asyncpg.connect(db_url)
|
||||
|
||||
# Generate a dummy embedding vector (384 dimensions for bge-small-en-v1.5)
|
||||
dummy_embedding = str([0.1] * 384)
|
||||
|
||||
print("=" * 80)
|
||||
print("PROFILING: Entry Points Query (Vector Similarity)")
|
||||
print("=" * 80)
|
||||
|
||||
# Run EXPLAIN ANALYZE
|
||||
explain = await conn.fetch("""
|
||||
EXPLAIN (ANALYZE, BUFFERS, VERBOSE)
|
||||
SELECT id, text, context, event_date, access_count, embedding,
|
||||
1 - (embedding <=> $1::vector) AS similarity
|
||||
FROM memory_units
|
||||
WHERE agent_id = $2
|
||||
AND embedding IS NOT NULL
|
||||
AND (1 - (embedding <=> $1::vector)) >= 0.5
|
||||
ORDER BY embedding <=> $1::vector
|
||||
LIMIT 3
|
||||
""", dummy_embedding, "test_agent")
|
||||
|
||||
for row in explain:
|
||||
print(row[0])
|
||||
|
||||
await conn.close()
|
||||
|
||||
|
||||
async def profile_neighbors_query(sample_node_ids):
|
||||
"""Profile the neighbors JOIN query."""
|
||||
db_url = os.getenv("DATABASE_URL")
|
||||
conn = await asyncpg.connect(db_url)
|
||||
|
||||
print("\n" + "=" * 80)
|
||||
print("PROFILING: Neighbors Query (Graph Traversal)")
|
||||
print(f"Sample size: {len(sample_node_ids)} nodes")
|
||||
print("=" * 80)
|
||||
|
||||
# Run EXPLAIN ANALYZE
|
||||
explain = await conn.fetch("""
|
||||
EXPLAIN (ANALYZE, BUFFERS, VERBOSE)
|
||||
SELECT ml.from_unit_id, ml.to_unit_id, ml.weight, ml.link_type, ml.entity_id,
|
||||
mu.text, mu.context, mu.event_date, mu.access_count,
|
||||
mu.id as neighbor_id
|
||||
FROM memory_links ml
|
||||
JOIN memory_units mu ON ml.to_unit_id = mu.id
|
||||
WHERE ml.from_unit_id = ANY($1::uuid[])
|
||||
AND ml.weight >= 0.1
|
||||
ORDER BY ml.from_unit_id, ml.weight DESC
|
||||
""", sample_node_ids)
|
||||
|
||||
for row in explain:
|
||||
print(row[0])
|
||||
|
||||
await conn.close()
|
||||
|
||||
|
||||
async def profile_embeddings_query(sample_node_ids):
|
||||
"""Profile the batch embeddings fetch query."""
|
||||
db_url = os.getenv("DATABASE_URL")
|
||||
conn = await asyncpg.connect(db_url)
|
||||
|
||||
print("\n" + "=" * 80)
|
||||
print("PROFILING: Embeddings Query (Batch Fetch)")
|
||||
print(f"Sample size: {len(sample_node_ids)} nodes")
|
||||
print("=" * 80)
|
||||
|
||||
# Run EXPLAIN ANALYZE
|
||||
explain = await conn.fetch("""
|
||||
EXPLAIN (ANALYZE, BUFFERS, VERBOSE)
|
||||
SELECT id, embedding
|
||||
FROM memory_units
|
||||
WHERE id = ANY($1::uuid[])
|
||||
""", sample_node_ids)
|
||||
|
||||
for row in explain:
|
||||
print(row[0])
|
||||
|
||||
await conn.close()
|
||||
|
||||
|
||||
async def get_sample_node_ids(batch_size=50):
|
||||
"""Get sample node IDs for profiling."""
|
||||
db_url = os.getenv("DATABASE_URL")
|
||||
conn = await asyncpg.connect(db_url)
|
||||
|
||||
rows = await conn.fetch(f"""
|
||||
SELECT id FROM memory_units
|
||||
LIMIT {batch_size}
|
||||
""")
|
||||
|
||||
await conn.close()
|
||||
return [row['id'] for row in rows]
|
||||
|
||||
|
||||
async def check_indexes():
|
||||
"""Check what indexes exist."""
|
||||
db_url = os.getenv("DATABASE_URL")
|
||||
conn = await asyncpg.connect(db_url)
|
||||
|
||||
print("\n" + "=" * 80)
|
||||
print("CURRENT INDEXES")
|
||||
print("=" * 80)
|
||||
|
||||
indexes = await conn.fetch("""
|
||||
SELECT
|
||||
tablename,
|
||||
indexname,
|
||||
indexdef
|
||||
FROM pg_indexes
|
||||
WHERE schemaname = 'public'
|
||||
AND tablename IN ('memory_units', 'memory_links')
|
||||
ORDER BY tablename, indexname
|
||||
""")
|
||||
|
||||
for idx in indexes:
|
||||
print(f"\nTable: {idx['tablename']}")
|
||||
print(f"Index: {idx['indexname']}")
|
||||
print(f"Definition: {idx['indexdef']}")
|
||||
|
||||
await conn.close()
|
||||
|
||||
|
||||
async def main():
|
||||
print("Starting Query Profiling...")
|
||||
|
||||
# Check indexes first
|
||||
await check_indexes()
|
||||
|
||||
# Get sample node IDs
|
||||
sample_ids = await get_sample_node_ids(50)
|
||||
|
||||
if sample_ids:
|
||||
print(f"\nGot {len(sample_ids)} sample node IDs for profiling")
|
||||
|
||||
# Profile each query type
|
||||
await profile_entry_points_query()
|
||||
await profile_neighbors_query(sample_ids)
|
||||
await profile_embeddings_query(sample_ids)
|
||||
else:
|
||||
print("\nNo data in database - run ingestion first")
|
||||
|
||||
print("\n" + "=" * 80)
|
||||
print("PROFILING COMPLETE")
|
||||
print("=" * 80)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
Loading…
Reference in a new issue