diff --git a/benchmarks/locomo/benchmark_results_think.json b/benchmarks/locomo/benchmark_results_think.json new file mode 100644 index 00000000..8137521c --- /dev/null +++ b/benchmarks/locomo/benchmark_results_think.json @@ -0,0 +1,356 @@ +{ + "overall_accuracy": 100.0, + "total_correct": 1, + "total_questions": 1, + "num_items": 1, + "item_results": [ + { + "item_id": "conv-26", + "metrics": { + "accuracy": 100.0, + "correct": 1, + "total": 1, + "category_stats": { + "2": { + "correct": 1, + "total": 1 + } + }, + "detailed_results": [ + { + "question": "When did Caroline go to the LGBTQ support group?", + "correct_answer": "7 May 2023", + "predicted_answer": "Caroline attended the LGBTQ support group on May\u202f7\u202f2023 (event recorded at 2023\u201105\u201107T13:56:00\u202fUTC).", + "reasoning": "Think API: 20 world facts, 0 agent facts, 20 opinions", + "category": 2, + "retrieved_memories": [ + { + "id": "5ddabf72-967c-4bd0-a0a8-1545324e48f2", + "text": "Caroline attended an LGBTQ support group on 2023-05-07 and found it powerful.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_1)", + "event_date": "2023-05-07T13:56:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "7b85f829-743e-4827-8349-ca6d01dba5dc", + "text": "Caroline volunteered at an LGBTQ+ youth center", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_15)", + "event_date": "2023-08-28T15:19:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "6794d0e8-5e4d-44bb-8238-21249021d101", + "text": "Caroline joined the LGBTQ activist group \"Connected LGBTQ Activists\" on 2023-07-18.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_10)", + "event_date": "2023-07-18T20:56:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "eb8952a1-9b7d-45b3-8f63-2411d6421725", + "text": "Caroline tried to apologize to the people she encountered during the hike.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_14)", + "event_date": "2023-08-18T13:33:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "2decc740-9674-4b25-acf1-1da8ba8f61dd", + "text": "Caroline owns a guinea pig named Oscar.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_13)", + "event_date": "2023-08-23T15:31:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "1aad5731-397b-4c3f-b78d-58768d376133", + "text": "Caroline told Melanie that she is lucky to have such an awesome family.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_10)", + "event_date": "2023-07-20T20:56:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "5b3e87cc-6892-4fb2-945e-5921a32c3453", + "text": "Caroline attended a poetry reading on Friday, October 6, 2023.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_17)", + "event_date": "2023-10-06T10:31:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "3fd7f709-6b78-443c-8768-abeee8f28e31", + "text": "Caroline created a self-portrait last week and posted a photo of it.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_13)", + "event_date": "2023-08-16T15:31:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "640e4293-755b-4634-b570-1f15128eb1e6", + "text": "The event room was electric with energy and support, and the posters displayed pride and strength, which inspired Caroline to create new artwork.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_17)", + "event_date": "2023-10-06T10:31:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "a9331948-bf5f-4791-86e9-0bf62342ed88", + "text": "Caroline visited the beach.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_14)", + "event_date": "2023-08-18T13:33:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "c48b29f0-bcb4-4fde-b616-4e08d9166146", + "text": "Caroline attended an adoption advice and assistance group, receiving a lot of help.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_13)", + "event_date": "2023-08-23T15:31:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "eba11741-e2ae-4ccf-976a-4df864b0cace", + "text": "Caroline heard transgender stories at the LGBTQ support group, which she found inspiring.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_1)", + "event_date": "2023-05-07T13:56:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "5968eee3-380d-4780-afec-9576fbafe422", + "text": "Caroline is interested in a career in counseling or mental health to support people with similar issues.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_1)", + "event_date": "2023-05-08T13:56:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "aee0214f-e49d-47a4-83f5-5f97304a3a51", + "text": "Caroline feels supported by people around her, which makes her feel okay.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_12)", + "event_date": "2023-08-17T13:50:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "149347b4-b7cb-4bf9-a8be-f743700a9783", + "text": "The city held a pride parade on 2023-07-15, where many people marched, waved flags, held signs, and celebrated love and diversity; Caroline missed the parade.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_10)", + "event_date": "2023-07-15T20:56:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "21ac438b-ce0b-4efd-8c6e-d424295ec3cc", + "text": "Caroline created a recent painting that represents inclusivity and diversity and uses it to speak up for the LGBTQ+ community and push for acceptance.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_11)", + "event_date": "2023-08-14T14:24:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "442b4482-f31b-4ef5-b545-eb0befab7cfe", + "text": "Caroline mentioned an advocacy event that was a cool experience with love and support.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_11)", + "event_date": "2023-08-14T14:24:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "a100f54b-7112-4616-8f7c-9a7347587597", + "text": "Caroline had a not-so-great experience on a hike where she ran into a group of religious conservatives who said something that upset her, leading her to reflect on the need for more work on LGBTQ rights.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_12)", + "event_date": "2023-08-17T13:50:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "3e2e6418-29db-44e4-963f-ea2089d33dd5", + "text": "Caroline has a lifelong love for horses.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_13)", + "event_date": "2023-08-23T15:31:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "0dda577c-9ea3-4a47-835a-63cade830838", + "text": "Caroline gave a talk at her school event last week about her transgender journey, encouraging students to get involved in the LGBTQ community and observed positive reactions from the audience.", + "context": "Conversation session between Caroline and Melanie (conversation conv-26 session session_3)", + "event_date": "2023-06-02T19:55:00+00:00", + "score": 0.0, + "fact_type": "world" + }, + { + "id": "054d8d37-aa7d-4a21-b4bd-7968e29d74dd", + "text": "Caroline is a participant within the LGBTQ community", + "context": "formed during thinking about: In what ways is Caroline participating in the LGBTQ community?", + "event_date": "2025-11-04T18:24:51+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "9862a174-0e48-418c-b796-2cb1185b7f61", + "text": "Caroline's plan to build a personal library for her future children suggests she may consider formal study in library science or related fields.", + "context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?", + "event_date": "2025-11-04T15:30:15+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "7318bff4-0702-44b9-8aa6-b50f5a59be86", + "text": "We cannot determine whether Caroline wants to move back to her home country soon", + "context": "formed during thinking about: Would Caroline want to move back to her home country soon?", + "event_date": "2025-11-04T18:05:51+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "cdb11bda-7c70-4b61-b36e-bdf21727a194", + "text": "Caroline is an event organizer within the LGBTQ community", + "context": "formed during thinking about: In what ways is Caroline participating in the LGBTQ community?", + "event_date": "2025-11-04T18:24:51+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "1d68207b-03e2-44ca-a79b-47ba427c1ba1", + "text": "Melanie and Caroline describe their journey as collaborative, supportive, and inspiring", + "context": "formed during thinking about: How do Melanie and Caroline describe their journey through life together?", + "event_date": "2025-11-04T18:24:08+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "8b6a13ff-7417-4458-a19c-cd101d24c53f", + "text": "Caroline has passed adoption agency interviews.", + "context": "formed during thinking about: What is Caroline's identity?", + "event_date": "2025-11-04T15:30:19+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "4e8a04ef-ea96-48ec-b016-c32a111bc819", + "text": "Caroline is an advocate within the LGBTQ community", + "context": "formed during thinking about: In what ways is Caroline participating in the LGBTQ community?", + "event_date": "2025-11-04T18:24:51+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "87c97473-6596-4e5c-bd0f-f989ab76e137", + "text": "The provided facts only describe Caroline's activities, goals, and interests in her current location", + "context": "formed during thinking about: Would Caroline want to move back to her home country soon?", + "event_date": "2025-11-04T18:05:51+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "6aa8c0fd-dffc-4d27-9559-b0687d263f59", + "text": "Caroline plays piano.", + "context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?", + "event_date": "2025-11-04T15:27:06+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "04372c16-57ae-429b-8f9c-e3ab9ef3bacd", + "text": "Caroline could integrate art therapy classes into counseling work.", + "context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?", + "event_date": "2025-11-04T15:27:06+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "d596524c-fdd6-4080-b7a7-f9da2cd3d6ed", + "text": "Caroline explicitly said she wants to continue her education.", + "context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?", + "event_date": "2025-11-04T18:08:28+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "f5308bcf-a59e-4b1d-90e0-0e243b79b98d", + "text": "Caroline paints.", + "context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?", + "event_date": "2025-11-04T15:27:06+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "f97a5f20-1e8d-45bc-93db-6626070ca5f1", + "text": "Caroline is interested in a career in mental-health services.", + "context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?", + "event_date": "2025-11-04T18:08:28+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "fc805a93-3b1d-4f8a-a98e-3b7044bcb2d0", + "text": "Caroline feels a sense of empowerment from activism.", + "context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?", + "event_date": "2025-11-04T15:27:06+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "31c26b9a-f4a3-4a08-bae8-410780b40414", + "text": "Previously I thought Caroline would likely focus on LGBTQ+ studies, gender studies, or social justice based on her activism. However, her explicit statements about pursuing a career in counseling and mental health, and her desire to support people with similar issues, indicate that her primary focus is now likely on a counseling or mental\u2011health career, potentially integrating her advocacy for marginalized communities.", + "context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?", + "event_date": "2025-11-04T15:27:06+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "987325a5-6b67-42bf-8910-fd194987bd35", + "text": "Caroline is a future parent-to-be.", + "context": "formed during thinking about: What is Caroline's identity?", + "event_date": "2025-11-04T15:30:19+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "ba5ae772-2d00-4cbf-ac88-6988bca1be78", + "text": "Melanie and Caroline frame challenges as a shared roller-coaster", + "context": "formed during thinking about: How do Melanie and Caroline describe their journey through life together?", + "event_date": "2025-11-04T18:24:08+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "de3b98e6-f435-41b7-a38b-712a35cd0c54", + "text": "There is no information indicating Caroline identifies as religious.", + "context": "formed during thinking about: Would Caroline be considered religious?", + "event_date": "2025-11-04T18:05:26+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "314d9820-e75d-455f-bfad-55dcea90c044", + "text": "Caroline may choose a multidisciplinary program that combines counseling/mental\u2011health training, LGBTQ+ advocacy, and early childhood literacy.", + "context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?", + "event_date": "2025-11-04T15:30:15+00:00", + "score": 0.0, + "fact_type": "opinion" + }, + { + "id": "5ba2cc08-4998-4667-8b95-adb1f0986504", + "text": "Caroline is most likely to pursue LGBTQ+ Studies and Gender & Social Justice.", + "context": "formed during thinking about: What fields would Caroline be likely to pursue in her educaton?", + "event_date": "2025-11-04T15:30:15+00:00", + "score": 0.0, + "fact_type": "opinion" + } + ], + "is_correct": true, + "correctness_reasoning": "The generated answer states the same date (May\u202f7\u202f2023) as the gold answer, so it is correct." + } + ] + }, + "num_sessions": -1 + } + ] +} \ No newline at end of file diff --git a/benchmarks/locomo/results_table_think.md b/benchmarks/locomo/results_table_think.md new file mode 100644 index 00000000..764c18dd --- /dev/null +++ b/benchmarks/locomo/results_table_think.md @@ -0,0 +1,7 @@ +# LoComo Benchmark Results (Think Mode) + +**Overall Accuracy**: 100.00% (1/1) + +| Sample ID | Sessions | Questions | Correct | Accuracy | Multi-hop | Single-hop | Temporal | Open-domain | +|-----------|----------|-----------|---------|----------|-----------|------------|----------|-------------| +| conv-26 | -1 | 1 | 1 | 100.00% | N/A | N/A | N/A | N/A | \ No newline at end of file diff --git a/benchmarks/visualizer/README.md b/benchmarks/visualizer/README.md new file mode 100644 index 00000000..e938c79a --- /dev/null +++ b/benchmarks/visualizer/README.md @@ -0,0 +1,62 @@ +# Benchmark Visualizer + +A standalone web service for visualizing benchmark results. Currently supports the LoComo benchmark with plans to add more benchmarks in the future. + +## Features + +- Interactive web interface for viewing benchmark results +- Detailed breakdown by category (Multi-hop, Single-hop, Temporal, Open-domain) +- Filter options to view all, correct, or incorrect answers +- Expandable Q&A details with reasoning and retrieved memories +- Overall and per-item accuracy statistics + +## Running the Visualizer + +### Option 1: Using the serve script (recommended) + +```bash +cd benchmarks/visualizer +./serve.sh +``` + +### Option 2: Using uvicorn directly + +```bash +cd benchmarks/visualizer +uv run uvicorn server:app --reload --host 0.0.0.0 --port 8001 +``` + +Then open your browser to: http://localhost:8001 + +## Usage + +1. Select a benchmark from the dropdown: + - **LoComo (search)**: Traditional two-step approach (search → LLM answer generation) + - **LoComo (think)**: Integrated approach using think API (single call for retrieval + reasoning) +2. The visualization will automatically load and display: + - Overall accuracy statistics + - Category-wise performance breakdown + - Detailed results for each conversation +3. Use the filter controls to show all answers, only incorrect, or only correct answers +4. Expand individual conversations to see Q&A details, reasoning, and retrieved memories + +## API Endpoints + +- `GET /` - Main visualizer page +- `GET /api/locomo?mode={search|think}` - Returns LoComo benchmark results as JSON + - `mode=search` (default): Returns results from `benchmark_results.json` + - `mode=think`: Returns results from `benchmark_results_think.json` + +## Requirements + +- FastAPI +- Uvicorn +- Python 3.11+ + +The visualizer reads benchmark results from: +- `benchmarks/locomo/benchmark_results.json` for search mode +- `benchmarks/locomo/benchmark_results_think.json` for think mode + +Make sure to run the benchmark first to generate results: +- Search mode: `cd benchmarks/locomo && uv run python run_benchmark.py` +- Think mode: `cd benchmarks/locomo && uv run python run_benchmark.py --use-think` diff --git a/benchmarks/visualizer/serve.sh b/benchmarks/visualizer/serve.sh new file mode 100755 index 00000000..760b54c5 --- /dev/null +++ b/benchmarks/visualizer/serve.sh @@ -0,0 +1,4 @@ +#!/bin/bash +# Start the Benchmark Visualizer server with hot reload +cd "$(dirname "$0")" +uv run uvicorn server:app --reload --host 0.0.0.0 --port 8001 diff --git a/benchmarks/visualizer/server.py b/benchmarks/visualizer/server.py new file mode 100644 index 00000000..245bb14a --- /dev/null +++ b/benchmarks/visualizer/server.py @@ -0,0 +1,72 @@ +"""Benchmark Visualizer Web Service. + +A standalone web service for visualizing benchmark results. +Currently supports LoComo benchmark visualization. +""" + +import json +from pathlib import Path +from typing import Any + +from fastapi import FastAPI, HTTPException +from fastapi.responses import HTMLResponse, FileResponse +from fastapi.staticfiles import StaticFiles + +app = FastAPI(title="Benchmark Visualizer") + +# Get the benchmarks directory +BENCHMARKS_DIR = Path(__file__).parent.parent + + +@app.get("/", response_class=HTMLResponse) +async def index(): + """Serve the main benchmark visualizer page.""" + html_path = Path(__file__).parent / "static" / "index.html" + with open(html_path) as f: + return f.read() + + +@app.get("/api/locomo") +async def get_locomo_results(mode: str = "search") -> dict[str, Any]: + """Get LoComo benchmark results. + + Returns pre-computed benchmark results from the locomo directory. + + Args: + mode: Either "search" (default) or "think" to select which results to load + """ + try: + # Determine filename based on mode + if mode == "think": + filename = "benchmark_results_think.json" + else: + filename = "benchmark_results.json" + + results_path = BENCHMARKS_DIR / "locomo" / filename + + if not results_path.exists(): + raise HTTPException( + status_code=404, + detail=f"Benchmark results not found for mode '{mode}'. Please run the benchmark first with {'--use-think' if mode == 'think' else 'default settings'}." + ) + + with open(results_path) as f: + results = json.load(f) + + return results + except json.JSONDecodeError as e: + raise HTTPException( + status_code=500, + detail=f"Failed to parse benchmark results: {str(e)}" + ) + except Exception as e: + raise HTTPException(status_code=500, detail=str(e)) + + +# Mount static files +app.mount("/static", StaticFiles(directory=Path(__file__).parent / "static"), name="static") + + +if __name__ == "__main__": + import uvicorn + uvicorn.run(app, host="127.0.0.1", port=8001) diff --git a/benchmarks/visualizer/static/css/styles.css b/benchmarks/visualizer/static/css/styles.css new file mode 100644 index 00000000..74664e4e --- /dev/null +++ b/benchmarks/visualizer/static/css/styles.css @@ -0,0 +1,140 @@ +body { + font-family: Tahoma, sans-serif; + margin: 0; + padding: 0; + background: #f5f5f5; +} + +.header { + background: #333; + color: white; + padding: 20px; + border-bottom: 3px solid #42a5f5; +} + +.header h1 { + margin: 0 0 5px 0; +} + +.header p { + margin: 0; + color: #ccc; +} + +.benchmark-selector { + background: #f0f0f0; + padding: 15px 20px; + border-bottom: 2px solid #333; + display: flex; + gap: 15px; + align-items: center; +} + +.benchmark-selector label { + font-weight: bold; + font-size: 14px; +} + +.benchmark-selector select { + padding: 8px 12px; + border: 2px solid #42a5f5; + border-radius: 4px; + background: white; + color: #333; + font-size: 14px; + font-weight: bold; + cursor: pointer; + min-width: 200px; +} + +#benchmark-content { + padding: 20px; +} + +.welcome-message { + text-align: center; + padding: 60px 20px; + color: #666; +} + +.welcome-message h2 { + color: #333; +} + +.load-button { + padding: 8px 20px; + background: #66bb6a; + color: white; + border: none; + border-radius: 4px; + cursor: pointer; + font-weight: bold; + font-size: 14px; + margin-bottom: 20px; +} + +.load-button:hover { + background: #43a047; +} + +.error-message { + color: #d32f2f; + padding: 20px; + background: #ffebee; + border: 2px solid #ef5350; + border-radius: 8px; + margin: 20px; + max-width: 800px; +} + +.error-message h3 { + margin-top: 0; + color: #c62828; +} + +.error-message pre { + background: #f5f5f5; + padding: 10px; + border-radius: 4px; + overflow-x: auto; + color: #333; + font-family: monospace; + font-size: 13px; +} + +.stats-grid { + display: grid; + grid-template-columns: repeat(auto-fit, minmax(200px, 1fr)); + gap: 10px; + margin-top: 10px; +} + +.stat-item { + padding: 12px; + background: white; + border: 1px solid #ddd; + border-radius: 4px; + text-align: center; +} + +.stat-label { + font-weight: bold; + color: #666; + font-size: 12px; + margin-bottom: 5px; +} + +.stat-value { + font-size: 24px; + color: #333; + font-weight: bold; +} + +.qa-results { + margin-top: 20px; +} + +.qa-item { + margin-bottom: 15px; + border-radius: 8px; +} diff --git a/benchmarks/visualizer/static/index.html b/benchmarks/visualizer/static/index.html new file mode 100644 index 00000000..23e1b5a3 --- /dev/null +++ b/benchmarks/visualizer/static/index.html @@ -0,0 +1,32 @@ + + + + Benchmark Visualizer + + + + +
+

Benchmark Visualizer

+

Analyze and visualize benchmark results

+
+ +
+ + +
+ +
+
+

Welcome to Benchmark Visualizer

+

Select a benchmark from the dropdown above to view results.

+
+
+ + + + diff --git a/benchmarks/visualizer/static/js/app.js b/benchmarks/visualizer/static/js/app.js new file mode 100644 index 00000000..71733f27 --- /dev/null +++ b/benchmarks/visualizer/static/js/app.js @@ -0,0 +1,342 @@ +// Benchmark Visualizer App + +let currentBenchmark = null; +let benchmarkData = null; + +function selectBenchmark() { + const select = document.getElementById('benchmark-select'); + currentBenchmark = select.value; + + if (!currentBenchmark) { + document.getElementById('benchmark-content').innerHTML = ` +
+

Welcome to Benchmark Visualizer

+

Select a benchmark from the dropdown above to view results.

+
+ `; + return; + } + + // Load the selected benchmark + if (currentBenchmark === 'locomo-search') { + loadLocomoResults('search'); + } else if (currentBenchmark === 'locomo-think') { + loadLocomoResults('think'); + } +} + +async function loadLocomoResults(mode = 'search') { + try { + const response = await fetch(`/api/locomo?mode=${mode}`); + + if (!response.ok) { + const errorData = await response.json(); + const modeLabel = mode === 'think' ? 'think' : 'search'; + const runCommand = mode === 'think' + ? 'uv run python run_benchmark.py --use-think' + : 'uv run python run_benchmark.py'; + + document.getElementById('benchmark-content').innerHTML = ` +
+

⚠️ Benchmark Results Not Found

+

${errorData.detail || 'The requested benchmark results are not available.'}

+

To generate ${modeLabel} mode results:

+
cd benchmarks/locomo
+${runCommand}
+

+ Once the benchmark completes, refresh this page and select "${mode === 'think' ? 'LoComo (think)' : 'LoComo (search)'}" again. +

+
+ `; + return; + } + + benchmarkData = await response.json(); + console.log(`Loaded locomo data (${mode} mode):`, benchmarkData); + renderLocomoResults(mode); + } catch (e) { + console.error('Error loading benchmark results:', e); + document.getElementById('benchmark-content').innerHTML = ` +
+

❌ Error Loading Results

+

${e.message}

+

Check the browser console for more details.

+
+ `; + } +} + +function renderLocomoResults(mode = 'search') { + if (!benchmarkData) return; + + const content = document.getElementById('benchmark-content'); + + try { + // Handle both old and new structure + const results = benchmarkData.item_results || benchmarkData.conversation_results || []; + const numItems = benchmarkData.num_items || results.length; + + console.log('Rendering results:', { resultsCount: results.length, numItems }); + + // Calculate per-category statistics + const categoryStats = { + 1: { name: 'Multi-hop', correct: 0, total: 0 }, + 2: { name: 'Single-hop', correct: 0, total: 0 }, + 3: { name: 'Temporal', correct: 0, total: 0 }, + 4: { name: 'Open-domain', correct: 0, total: 0 } + }; + + // Aggregate across all items + results.forEach(item => { + if (item.metrics && item.metrics.detailed_results) { + item.metrics.detailed_results.forEach(result => { + const category = result.category; + if (categoryStats[category]) { + categoryStats[category].total++; + if (result.is_correct) { + categoryStats[category].correct++; + } + } + }); + } + }); + + // Determine title based on mode + const modeLabel = mode === 'think' ? ' (Think Mode)' : ' (Search Mode)'; + + // Overall stats + const overallHtml = ` +
+

LoComo Benchmark${modeLabel} - Overall Performance

+
+
+
Overall Accuracy
+
${benchmarkData.overall_accuracy.toFixed(2)}%
+
+
+
Correct Answers
+
${benchmarkData.total_correct} / ${benchmarkData.total_questions}
+
+
+
Items
+
${numItems}
+
+
+ +

Accuracy by Category

+
+ ${Object.values(categoryStats).map(cat => { + const accuracy = cat.total > 0 ? ((cat.correct / cat.total) * 100).toFixed(1) : 0; + const color = accuracy >= 70 ? '#43a047' : accuracy >= 50 ? '#ff9800' : '#e53935'; + return ` +
+
${cat.name}
+
${accuracy}%
+
${cat.correct} / ${cat.total}
+
+ `; + }).join('')} +
+
+ `; + + // Filter controls + const filterHtml = ` +
+ + + + +
+ `; + + // Build item sections + let itemsHtml = ''; + results.forEach((item, idx) => { + const itemId = item.item_id || item.sample_id || `item-${idx}`; + const accuracy = item.metrics.accuracy.toFixed(2); + const correctCount = item.metrics.correct; + const totalCount = item.metrics.total; + + itemsHtml += ` +
+
+

+ 📊 ${itemId} + + ${accuracy}% (${correctCount}/${totalCount}) + +

+
+ +
+ `; + }); + + content.innerHTML = overallHtml + filterHtml + itemsHtml; + } catch (e) { + console.error('Error rendering Locomo results:', e); + content.innerHTML = ` +
+ Error rendering results: ${e.message}
+
${e.stack}
+
+ `; + } +} + +function renderConversationDetails(conv) { + if (!conv || !conv.metrics) { + return '
No metrics available
'; + } + + const results = conv.metrics.detailed_results; + if (!results || !Array.isArray(results) || results.length === 0) { + return '
No detailed results available
'; + } + + let html = '
'; + + results.forEach((result, idx) => { + const isCorrect = result.is_correct; + const bgColor = isCorrect ? '#e8f5e9' : '#ffebee'; + const icon = isCorrect ? '✅' : '❌'; + const category = getCategoryName(result.category); + + html += ` +
+
+
+
+ ${icon} Question ${idx + 1} ${category} +
+
+ Q: ${result.question} +
+
+
+ +
+
+
✓ Correct Answer:
+
+ ${result.correct_answer} +
+
+
+
+ ${isCorrect ? '✓' : '✗'} Predicted Answer: +
+
+ ${result.predicted_answer} +
+
+
+ +
+ + 📝 Show Reasoning & Retrieved Memories + +
+
+ System Reasoning: +
+ ${result.reasoning} +
+
+
+ Judge Reasoning: +
+ ${result.correctness_reasoning || 'N/A'} +
+
+
+ Retrieved Memories (${result.retrieved_memories ? result.retrieved_memories.length : 0}): + ${renderRetrievedMemories(result.retrieved_memories)} +
+
+
+
+ `; + }); + + html += '
'; + return html; +} + +function renderRetrievedMemories(memories) { + if (!memories || !Array.isArray(memories) || memories.length === 0) { + return '
No memories retrieved
'; + } + + let html = '
'; + memories.forEach((mem, idx) => { + if (!mem) return; + const eventDate = mem.event_date ? new Date(mem.event_date).toLocaleString() : 'N/A'; + + // Determine border color based on fact type + let borderColor = '#42a5f5'; // default blue + let factTypeLabel = ''; + if (mem.fact_type) { + factTypeLabel = `${mem.fact_type.toUpperCase()}`; + if (mem.fact_type === 'world') { + borderColor = '#4caf50'; // green + } else if (mem.fact_type === 'agent') { + borderColor = '#ff9800'; // orange + } else if (mem.fact_type === 'opinion') { + borderColor = '#9c27b0'; // purple + } + } + + html += ` +
+
+ Rank #${idx + 1} | Score: ${mem.score ? mem.score.toFixed(4) : 'N/A'} | Event Date: ${eventDate}${factTypeLabel} +
+
${mem.text}
+
+ `; + }); + html += '
'; + return html; +} + +function getCategoryName(category) { + const categories = { + 1: 'Multi-hop', + 2: 'Single-hop', + 3: 'Temporal', + 4: 'Open-domain' + }; + return categories[category] || 'Unknown'; +} + +function toggleConversation(idx) { + const elem = document.getElementById(`conv-${idx}`); + if (elem.style.display === 'none') { + elem.style.display = 'block'; + } else { + elem.style.display = 'none'; + } +} + +function filterAnswers() { + const filter = document.querySelector('input[name="answer-filter"]:checked').value; + const items = document.querySelectorAll('.qa-item'); + + items.forEach(item => { + const isCorrect = item.dataset.correct === 'true'; + + if (filter === 'all') { + item.style.display = 'block'; + } else if (filter === 'correct' && isCorrect) { + item.style.display = 'block'; + } else if (filter === 'incorrect' && !isCorrect) { + item.style.display = 'block'; + } else { + item.style.display = 'none'; + } + }); +} diff --git a/examples/parallel_think.py b/examples/parallel_think.py new file mode 100644 index 00000000..2f095302 --- /dev/null +++ b/examples/parallel_think.py @@ -0,0 +1,46 @@ +""" +Example: Running many think operations in parallel with optimized connection pooling. + +For 100 parallel think operations: +- Each think does 3 searches (world, agent, opinion) +- Each search acquires 1-3 connections briefly +- Total: ~300 concurrent connection requests + +Solution: Increase pool_max_size to handle the concurrency. +""" +import asyncio +from memora import TemporalSemanticMemory + + +async def main(): + # For 100 parallel think operations, use a larger pool + # Rule of thumb: pool_max_size >= (num_parallel_thinks * 3) + memory = TemporalSemanticMemory( + pool_min_size=10, # Keep some connections warm + pool_max_size=200 # Allow up to 200 concurrent connections + ) + await memory.initialize() + + # Example: Run 100 think operations in parallel + queries = [f"Query {i}" for i in range(100)] + + tasks = [ + memory.think_async( + agent_id="test_agent", + query=query, + thinking_budget=50, + top_k=10 + ) + for query in queries + ] + + # Run all thinks in parallel + results = await asyncio.gather(*tasks) + + print(f"Completed {len(results)} think operations") + + await memory.close() + + +if __name__ == "__main__": + asyncio.run(main()) diff --git a/scripts/profile_queries.py b/scripts/profile_queries.py new file mode 100644 index 00000000..e8a543c8 --- /dev/null +++ b/scripts/profile_queries.py @@ -0,0 +1,166 @@ +""" +Profile slow database queries to identify optimization opportunities. + +Usage: + uv run python scripts/profile_queries.py +""" +import asyncio +import os +import asyncpg +from dotenv import load_dotenv + +load_dotenv() + + +async def profile_entry_points_query(): + """Profile the vector similarity entry points query.""" + db_url = os.getenv("DATABASE_URL") + conn = await asyncpg.connect(db_url) + + # Generate a dummy embedding vector (384 dimensions for bge-small-en-v1.5) + dummy_embedding = str([0.1] * 384) + + print("=" * 80) + print("PROFILING: Entry Points Query (Vector Similarity)") + print("=" * 80) + + # Run EXPLAIN ANALYZE + explain = await conn.fetch(""" + EXPLAIN (ANALYZE, BUFFERS, VERBOSE) + SELECT id, text, context, event_date, access_count, embedding, + 1 - (embedding <=> $1::vector) AS similarity + FROM memory_units + WHERE agent_id = $2 + AND embedding IS NOT NULL + AND (1 - (embedding <=> $1::vector)) >= 0.5 + ORDER BY embedding <=> $1::vector + LIMIT 3 + """, dummy_embedding, "test_agent") + + for row in explain: + print(row[0]) + + await conn.close() + + +async def profile_neighbors_query(sample_node_ids): + """Profile the neighbors JOIN query.""" + db_url = os.getenv("DATABASE_URL") + conn = await asyncpg.connect(db_url) + + print("\n" + "=" * 80) + print("PROFILING: Neighbors Query (Graph Traversal)") + print(f"Sample size: {len(sample_node_ids)} nodes") + print("=" * 80) + + # Run EXPLAIN ANALYZE + explain = await conn.fetch(""" + EXPLAIN (ANALYZE, BUFFERS, VERBOSE) + SELECT ml.from_unit_id, ml.to_unit_id, ml.weight, ml.link_type, ml.entity_id, + mu.text, mu.context, mu.event_date, mu.access_count, + mu.id as neighbor_id + FROM memory_links ml + JOIN memory_units mu ON ml.to_unit_id = mu.id + WHERE ml.from_unit_id = ANY($1::uuid[]) + AND ml.weight >= 0.1 + ORDER BY ml.from_unit_id, ml.weight DESC + """, sample_node_ids) + + for row in explain: + print(row[0]) + + await conn.close() + + +async def profile_embeddings_query(sample_node_ids): + """Profile the batch embeddings fetch query.""" + db_url = os.getenv("DATABASE_URL") + conn = await asyncpg.connect(db_url) + + print("\n" + "=" * 80) + print("PROFILING: Embeddings Query (Batch Fetch)") + print(f"Sample size: {len(sample_node_ids)} nodes") + print("=" * 80) + + # Run EXPLAIN ANALYZE + explain = await conn.fetch(""" + EXPLAIN (ANALYZE, BUFFERS, VERBOSE) + SELECT id, embedding + FROM memory_units + WHERE id = ANY($1::uuid[]) + """, sample_node_ids) + + for row in explain: + print(row[0]) + + await conn.close() + + +async def get_sample_node_ids(batch_size=50): + """Get sample node IDs for profiling.""" + db_url = os.getenv("DATABASE_URL") + conn = await asyncpg.connect(db_url) + + rows = await conn.fetch(f""" + SELECT id FROM memory_units + LIMIT {batch_size} + """) + + await conn.close() + return [row['id'] for row in rows] + + +async def check_indexes(): + """Check what indexes exist.""" + db_url = os.getenv("DATABASE_URL") + conn = await asyncpg.connect(db_url) + + print("\n" + "=" * 80) + print("CURRENT INDEXES") + print("=" * 80) + + indexes = await conn.fetch(""" + SELECT + tablename, + indexname, + indexdef + FROM pg_indexes + WHERE schemaname = 'public' + AND tablename IN ('memory_units', 'memory_links') + ORDER BY tablename, indexname + """) + + for idx in indexes: + print(f"\nTable: {idx['tablename']}") + print(f"Index: {idx['indexname']}") + print(f"Definition: {idx['indexdef']}") + + await conn.close() + + +async def main(): + print("Starting Query Profiling...") + + # Check indexes first + await check_indexes() + + # Get sample node IDs + sample_ids = await get_sample_node_ids(50) + + if sample_ids: + print(f"\nGot {len(sample_ids)} sample node IDs for profiling") + + # Profile each query type + await profile_entry_points_query() + await profile_neighbors_query(sample_ids) + await profile_embeddings_query(sample_ids) + else: + print("\nNo data in database - run ingestion first") + + print("\n" + "=" * 80) + print("PROFILING COMPLETE") + print("=" * 80) + + +if __name__ == "__main__": + asyncio.run(main())