Files
Netherwarlord 8f246fabbc feat: add plugin framework with UI contributions and suite selection
- Implemented PluginBlocks component to render various plugin UI elements.
- Created SuitePicker for selecting test suites and questions.
- Introduced usePluginUI hook for managing plugin UI state and manifest.
- Developed DocsPage for comprehensive plugin framework documentation.
- Added PluginPage to render pages declared by plugins based on the current path.
2026-07-28 03:00:55 -04:00

98 lines
2.6 KiB
Python

"""Bridge between the server's internals and the plugin API.
Plugins only ever see the frozen dataclasses in ``plugin_api``. This module does
the translation and owns the markdown rendering of grades, so nothing in the
plugin contract is coupled to how runs happen to be stored.
"""
from __future__ import annotations
import sys
from pathlib import Path
from typing import TYPE_CHECKING, Any, Dict, List, Optional
ROOT_DIR = Path(__file__).resolve().parent.parent
if str(ROOT_DIR) not in sys.path:
sys.path.insert(0, str(ROOT_DIR))
from plugin_api import Grade, GradeEntry, RunRecord, TestRecord # noqa: E402
from plugin_system import ( # noqa: E402
PluginManager,
collect_report_sections,
get_plugin_manager,
letter_grade,
render_grade_section,
)
if TYPE_CHECKING: # pragma: no cover
from .run_manager import RunState, TestOutcome
__all__ = [
"Grade",
"GradeEntry",
"PluginManager",
"RunRecord",
"TestRecord",
"build_run_record",
"build_test_record",
"collect_report_sections",
"get_plugin_manager",
"letter_grade",
"render_grade_section",
]
def build_test_record(
outcome: "TestOutcome",
*,
total: int,
prompt_text: str,
) -> TestRecord:
return TestRecord(
index=outcome.index,
total=total,
title=outcome.title,
filename=outcome.filename,
suite=outcome.suite,
prompt=prompt_text,
ok=outcome.status == "ok",
response=outcome.response,
error=outcome.error,
metrics=dict(outcome.metrics or {}),
elapsed=outcome.elapsed,
)
def build_run_record(
run: "RunState",
*,
prompts_by_id: Dict[str, str],
report_path: Optional[Path] = None,
) -> RunRecord:
# Keyed by qualified "<suite>/<file>" ID. Looking a prompt up by bare
# filename would hand a grader the wrong question's text whenever two
# suites in one run share a filename, which is the common case.
tests = [
build_test_record(
outcome,
total=run.total,
prompt_text=prompts_by_id.get(outcome.question_id, ""),
)
for outcome in run.outcomes
]
return RunRecord(
id=run.id,
provider=run.provider,
model_id=run.model_id,
model_label=run.model_label,
temperature=run.temperature,
total=run.total,
status=run.status,
started_at=run.started_at,
finished_at=run.finished_at,
report_path=report_path,
summary=run.summary(),
tests=tests,
grades={index: list(entries) for index, entries in run.grades.items()},
)