Skip to content
Merged
Show file tree
Hide file tree
Changes from 1 commit
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
37 changes: 37 additions & 0 deletions .github/workflows/tests.yml
Original file line number Diff line number Diff line change
Expand Up @@ -213,3 +213,40 @@ jobs:
--verbose \
--redact \
--exit-code 1

# ── Performance benchmarks: summary cache (issue #7) ───────────────────────
Comment thread
clean6378-max-it marked this conversation as resolved.
Outdated
benchmarks:
name: Performance benchmarks (gated)
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
persist-credentials: false

- name: Set up Python
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6.2.0
with:
python-version: "3.12"

- name: Install runtime + benchmark dependencies
run: |
python -m pip install --upgrade pip
python -m pip install -r requirements-lock.txt
python -m pip install 'pytest>=8,<9' 'pytest-benchmark>=4,<5'

- name: Run summary-cache benchmarks
run: >
python -m pytest tests/benchmarks/
--benchmark-only
--benchmark-json=benchmark-results.json
--benchmark-columns=min,max,mean,stddev,rounds
-o addopts=
Comment thread
clean6378-max-it marked this conversation as resolved.

- name: Regression gate
run: python scripts/check_benchmark_regression.py benchmark-results.json benchmarks/baselines.json

- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
if: always()
with:
name: benchmark-results
path: benchmark-results.json
2 changes: 2 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -44,3 +44,5 @@ Thumbs.db
htmlcov/
coverage.xml
.hypothesis/
benchmark-results.json
benchmarks/_raw.json
15 changes: 15 additions & 0 deletions benchmarks/baselines.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
{
"_note": "Gated means from local reference run with 1.5x slack (Windows dev host). Refresh from ubuntu-latest CI artifact after first green benchmark job.",
"updated": "2026-06-25T00:00:00Z",
"machine": "Windows",
"groups": {
"summary-cache": {
"test_summary_cache_hit": 8.91e-05,
"test_summary_cache_miss": 8.13e-05,
"test_fingerprint_workspace_entries[10]": 0.001708,
"test_fingerprint_workspace_entries[50]": 0.005457,
"test_fingerprint_workspace_entries[200]": 0.01715,
"test_summary_cache_round_trip": 0.001667
}
}
}
Comment thread
coderabbitai[bot] marked this conversation as resolved.
8 changes: 8 additions & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -31,10 +31,18 @@ desktop = ["pywebview>=5.0,<6"]
# Development tooling: testing + type checking.
dev = [
"pytest>=8,<9",
"pytest-benchmark>=4,<5",
"mypy>=1.10,<2",
"hypothesis>=6.100,<7",
]

[tool.pytest.ini_options]
addopts = "--benchmark-skip"
testpaths = ["tests"]
Comment thread
clean6378-max-it marked this conversation as resolved.
markers = [
"benchmark: performance benchmarks (pytest-benchmark)",
]
Comment thread
coderabbitai[bot] marked this conversation as resolved.

[project.scripts]
# Primary CLI: export Cursor chat histories to Markdown / zip.
# Usage: cursor-chat-export [--since all|last] [--out DIR] [--no-zip] [--help]
Expand Down
142 changes: 142 additions & 0 deletions scripts/check_benchmark_regression.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,142 @@
"""Compare pytest-benchmark JSON output against stored baselines."""

from __future__ import annotations

import argparse
import json
import sys
from pathlib import Path

THRESHOLD = 1.20


class BenchmarkDataError(ValueError):
"""Raised when benchmark JSON input is malformed or missing required fields."""


def load_results(results_path: str | Path) -> dict[str, float]:
path = Path(results_path)
try:
data = json.loads(path.read_text(encoding="utf-8"))
except OSError as exc:
raise BenchmarkDataError(f"cannot read {path}: {exc}") from exc
except json.JSONDecodeError as exc:
raise BenchmarkDataError(f"invalid JSON in {path}: {exc}") from exc
try:
benchmarks = data["benchmarks"]
except (KeyError, TypeError) as exc:
raise BenchmarkDataError(f"{path} missing top-level 'benchmarks' array") from exc
if not isinstance(benchmarks, list):
raise BenchmarkDataError(f"{path} 'benchmarks' must be an array")

results: dict[str, float] = {}
for index, entry in enumerate(benchmarks):
if not isinstance(entry, dict):
raise BenchmarkDataError(f"{path} benchmarks[{index}] must be an object")
try:
name = entry["name"]
mean = float(entry["stats"]["mean"])
except (KeyError, TypeError, ValueError) as exc:
raise BenchmarkDataError(
f"{path} benchmarks[{index}] missing 'name' or 'stats.mean'"
) from exc
name = str(name)
if name in results:
raise BenchmarkDataError(f"{path} duplicate benchmark name {name!r}")
results[name] = mean
return results


def load_baseline_means(baselines_path: str | Path) -> dict[str, float]:
path = Path(baselines_path)
try:
data = json.loads(path.read_text(encoding="utf-8"))
except OSError as exc:
raise BenchmarkDataError(f"cannot read {path}: {exc}") from exc
except json.JSONDecodeError as exc:
raise BenchmarkDataError(f"invalid JSON in {path}: {exc}") from exc
if not isinstance(data, dict):
raise BenchmarkDataError(f"{path} root value must be an object")

if "groups" not in data:
raise BenchmarkDataError(f"{path} missing required 'groups' key")
groups = data["groups"]
if not isinstance(groups, dict):
raise BenchmarkDataError(f"{path} 'groups' must be an object")

means: dict[str, float] = {}
for group_name, value in groups.items():
if not isinstance(value, dict):
continue
for name, mean in value.items():
name = str(name)
if name in means:
raise BenchmarkDataError(f"{path} duplicate benchmark name {name!r} across groups")
try:
means[name] = float(mean)
except (TypeError, ValueError) as exc:
raise BenchmarkDataError(
f"{path} groups[{group_name!r}][{name!r}] is not a numeric mean"
) from exc
return means


def check_regression(
results_path: str | Path,
baselines_path: str | Path,
*,
threshold: float = THRESHOLD,
) -> int:
"""Return 0 when within threshold; 1 when any gated benchmark regresses."""
flat = load_results(results_path)
baseline_means = load_baseline_means(baselines_path)

failures: list[str] = []
for name, base in baseline_means.items():
cur = flat.get(name)
if cur is None:
print(f"WARN: no current result for baseline {name!r}; skipping")
continue
if base == 0:
print(f"WARN: baseline for {name!r} is zero; skipping ratio check")
continue
ratio = cur / base
tag = "FAIL" if ratio > threshold else "ok"
print(f"[{tag}] {name}: {cur:.6f}s vs {base:.6f}s ({ratio:.2f}x)")
if ratio > threshold:
failures.append(name)

Comment thread
clean6378-max-it marked this conversation as resolved.
for name in flat:
if name not in baseline_means:
print(f"WARN: {name!r} has no baseline yet; not gated")

if failures:
print(f"\nREGRESSION: {len(failures)} benchmark(s) exceeded {threshold:.0%}")
return 1
return 0


def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("results_path", help="pytest-benchmark --benchmark-json output")
parser.add_argument("baselines_path", help="path to benchmarks/baselines.json")
parser.add_argument(
"--threshold",
type=float,
default=THRESHOLD,
help="fail when current mean exceeds baseline by more than this ratio (default: 1.20)",
)
args = parser.parse_args(argv)
try:
return check_regression(
args.results_path,
args.baselines_path,
threshold=args.threshold,
)
except BenchmarkDataError as exc:
print(f"ERROR: {exc}", file=sys.stderr)
return 2


if __name__ == "__main__":
sys.exit(main())
89 changes: 89 additions & 0 deletions tests/benchmarks/conftest.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,89 @@
"""Synthetic workspace trees for summary-cache performance benchmarks."""

from __future__ import annotations

import os
import sys
from pathlib import Path
from typing import Any

import pytest

REPO_ROOT = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
if REPO_ROOT not in sys.path:
sys.path.insert(0, REPO_ROOT)

from services import summary_cache # noqa: E402
from services.summary_cache import fingerprint_workspace_storage # noqa: E402


def make_workspace_entries(workspace_root: Path, count: int) -> list[dict[str, Any]]:
"""Build *count* synthetic workspace entries with on-disk state files."""
entries: list[dict[str, Any]] = []
for i in range(count):
name = f"ws_{i:04d}"
entry_dir = workspace_root / name
entry_dir.mkdir(parents=True, exist_ok=True)
(entry_dir / "state.vscdb").write_bytes(b"bench")
workspace_json = entry_dir / "workspace.json"
workspace_json.write_text('{"folder": "/bench"}', encoding="utf-8")
entries.append(
{
"name": name,
"workspaceJsonPath": str(workspace_json),
}
)
return entries


@pytest.fixture
def summary_cache_dir(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path:
"""Redirect summary-cache files to an isolated temp directory."""
cache_dir = tmp_path / "cache"
cache_dir.mkdir()
monkeypatch.setattr(summary_cache, "CACHE_DIR", cache_dir)
monkeypatch.setattr(summary_cache, "PROJECTS_CACHE_FILE", cache_dir / "projects.json")
monkeypatch.setattr(
summary_cache,
"COMPOSER_MAP_CACHE_FILE",
cache_dir / "composer-id-to-ws.json",
)
Comment thread
clean6378-max-it marked this conversation as resolved.
return cache_dir


@pytest.fixture
def sample_projects() -> list[dict[str, Any]]:
return [
{
"id": "ws_0000",
"name": "Bench Project",
"conversationCount": 3,
"lastModified": "2026-06-24T00:00:00Z",
}
]


@pytest.fixture
def synthetic_workspace(tmp_path: Path, request: pytest.FixtureRequest) -> tuple[str, list[dict[str, Any]]]:
"""Workspace path + entries. Parametrize via indirect ``workspace_entry_count``."""
count = getattr(request, "param", 10)
workspace_root = tmp_path / "workspaceStorage"
workspace_root.mkdir()
entries = make_workspace_entries(workspace_root, count)
return str(workspace_root), entries


@pytest.fixture
def workspace_fingerprint(synthetic_workspace: tuple[str, list[dict[str, Any]]]) -> dict[str, Any]:
workspace_path, entries = synthetic_workspace
return fingerprint_workspace_storage(
workspace_path,
entries,
global_db_path=None,
rules=[],
)


@pytest.fixture
def stale_fingerprint(workspace_fingerprint: dict[str, Any]) -> dict[str, Any]:
return {**workspace_fingerprint, "global_db_mtime_ns": 9_999_999_999}
73 changes: 73 additions & 0 deletions tests/benchmarks/test_summary_cache_bench.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,73 @@
"""pytest-benchmark coverage for services/summary_cache.py hot paths."""

from __future__ import annotations

from pathlib import Path
from typing import Any

import pytest

from services.summary_cache import (
fingerprint_workspace_storage,
get_cached_projects,
set_cached_projects,
)

@pytest.mark.benchmark(group="summary-cache")
def test_summary_cache_hit(
benchmark,
summary_cache_dir: Path,
workspace_fingerprint: dict[str, Any],
sample_projects: list[dict[str, Any]],
) -> None:
set_cached_projects(workspace_fingerprint, sample_projects, [])
benchmark(get_cached_projects, workspace_fingerprint)


@pytest.mark.benchmark(group="summary-cache")
def test_summary_cache_miss(
benchmark,
summary_cache_dir: Path,
workspace_fingerprint: dict[str, Any],
stale_fingerprint: dict[str, Any],
sample_projects: list[dict[str, Any]],
) -> None:
set_cached_projects(workspace_fingerprint, sample_projects, [])
benchmark(get_cached_projects, stale_fingerprint)
Comment thread
clean6378-max-it marked this conversation as resolved.


@pytest.mark.benchmark(group="summary-cache")
@pytest.mark.parametrize(
"synthetic_workspace",
[10, 50, 200],
indirect=True,
)
def test_fingerprint_workspace_entries(
benchmark,
synthetic_workspace: tuple[str, list[dict[str, Any]]],
) -> None:
workspace_path, entries = synthetic_workspace
benchmark(
fingerprint_workspace_storage,
workspace_path,
entries,
global_db_path=None,
rules=[],
)


@pytest.mark.benchmark(group="summary-cache")
def test_summary_cache_round_trip(
benchmark,
summary_cache_dir: Path,
workspace_fingerprint: dict[str, Any],
sample_projects: list[dict[str, Any]],
) -> None:
fp = workspace_fingerprint
projects = sample_projects

def _run() -> None:
set_cached_projects(fp, projects, [])
get_cached_projects(fp)

benchmark(_run)
Loading