From c9d1b86b1cbdd5019c0b6f200968e596b8bd5b41 Mon Sep 17 00:00:00 2001 From: invi-bhagyesh <104337027+invi-bhagyesh@users.noreply.github.com> Date: Thu, 24 Sep 2026 21:41:56 +0530 Subject: [PATCH 1/2] Batch evaluation presentation queries to avoid slow history loading --- backend/app/api.py | 22 +++++++++++++++------- backend/app/db.py | 12 ++++++++++++ backend/tests/test_backend.py | 25 +++++++++++++++++++++++++ 3 files changed, 52 insertions(+), 7 deletions(-) diff --git a/backend/app/api.py b/backend/app/api.py index 896756f7..80c69921 100644 --- a/backend/app/api.py +++ b/backend/app/api.py @@ -113,7 +113,9 @@ def admin_credit(member_id:UUID,incoming:CreditGrant,actor=Depends(administrator @app.get('/admin/evaluations') def admin_jobs(actor=Depends(administrator)): - return [present(j) | {'user_id':j['user_id']} for j in db.list()[:200]] + rows = db.list()[:200] + metadata = db.get_presentations([j['id'] for j in rows]) + return [present(j, metadata.get(j['id'], {})) | {'user_id':j['user_id']} for j in rows] @app.get('/admin/evaluations/{job_id}/logs') def admin_logs(job_id:UUID,actor=Depends(administrator)): @@ -145,15 +147,17 @@ def worker(job_id: UUID, authorization: str | None = Header(default=None)): if not job: raise HTTPException(404, 'Evaluation not found') return job - def present(job): - job = {**job, 'allocation': db.get_presentation(job['id'], 'allocation')} - publication=db.get_presentation(job['id'],'publication') + def present(job, metadata=None): + if metadata is None: + metadata = db.get_presentations([job['id']]).get(job['id'], {}) + job = {**job, 'allocation': metadata.get('allocation')} + publication=metadata.get('publication') if job['state']=='succeeded': desired=job['config'].get('visibility')=='public' if desired and not publication:publication={'state':'pending'} elif publication and publication.get('desired_public')!=desired: publication={**publication,'state':'pending' if desired else 'unpublishing'} - summary=db.get_presentation(job['id'],'summary') or {} + summary=metadata.get('summary') or {} return public_job(job) | {'publication':publication, 'omitted_count': summary.get('omitted_count',0)} @app.get('/health') @@ -240,11 +244,15 @@ def schema(): return EvaluationRequest.model_json_schema() @app.get('/evaluations') def evaluations(user_id=Depends(user)): - return [present(job) for job in db.list(user_id)] + rows = db.list(user_id) + metadata = db.get_presentations([j['id'] for j in rows]) + return [present(job, metadata.get(job['id'], {})) for job in rows] @app.get('/experiments') def published(): - return [present(j) for j in db.public_jobs() if db.get_presentation(j['id'], 'summary')] + rows = db.public_jobs() + metadata = db.get_presentations([j['id'] for j in rows]) + return [present(j, metadata.get(j['id'], {})) for j in rows if metadata.get(j['id'], {}).get('summary')] def readable(job_id, authorization): job = db.get(str(job_id)) diff --git a/backend/app/db.py b/backend/app/db.py index 552e0259..aa3d2726 100644 --- a/backend/app/db.py +++ b/backend/app/db.py @@ -185,6 +185,18 @@ def get_presentation(self, job_id, name): with self.engine.connect() as c: return c.execute(select(presentation.c.data).where(presentation.c.job_id == job_id, presentation.c.name == name)).scalar() + def get_presentations(self, job_ids): + if not job_ids: + return {} + with self.engine.connect() as c: + rows = c.execute(select(presentation).where( + presentation.c.job_id.in_(job_ids), + presentation.c.name.in_(('allocation', 'publication', 'summary')))).mappings() + result = {} + for row in rows: + result.setdefault(row['job_id'], {})[row['name']] = row['data'] + return result + def visibility(self, job_id, value): with self.engine.begin() as c: row = c.execute(select(jobs).where(jobs.c.id == job_id).with_for_update()).mappings().one() diff --git a/backend/tests/test_backend.py b/backend/tests/test_backend.py index 78588a52..4952a910 100644 --- a/backend/tests/test_backend.py +++ b/backend/tests/test_backend.py @@ -386,3 +386,28 @@ def test_auto_cpu_respects_cpu_limit(service): response=submit(client,payload(compute_type='gpu')) assert response.status_code==403 assert 'max_cpu_count' in response.json()['detail'] + + +def test_history_batches_metadata_and_keeps_owner_scope(service): + from sqlalchemy import event + _, db, client = service + first = submit(client).json()['id'] + second = submit(client, key='two').json()['id'] + foreign = submit(client, user=OTHER).json()['id'] + db.put_presentation(first, 'summary', {'omitted_count': 3}) + db.put_presentation(foreign, 'summary', {'omitted_count': 999}) + statements = [] + def capture(conn, cursor, statement, parameters, context, executemany): + if statement.lstrip().upper().startswith('SELECT') and 'va_presentation' in statement: + statements.append(statement) + event.listen(db.engine, 'before_cursor_execute', capture) + try: + response = client.get('/evaluations', headers={'Authorization': 'Bearer '+USER}) + finally: + event.remove(db.engine, 'before_cursor_execute', capture) + assert response.status_code == 200 + rows = {row['id']: row for row in response.json()} + assert set(rows) == {first, second} + assert rows[first]['omitted_count'] == 3 + assert rows[second]['omitted_count'] == 0 + assert len(statements) == 1 From 33f414440f495e727ff360948a59c0018e0313b6 Mon Sep 17 00:00:00 2001 From: invi-bhagyesh <104337027+invi-bhagyesh@users.noreply.github.com> Date: Thu, 24 Sep 2026 21:41:56 +0530 Subject: [PATCH 2/2] Add a styled successful runs table above evaluation history --- next-src/src/app/pages.css | 10 ++++++++++ next-src/src/components/EvaluationRunner.tsx | 2 +- 2 files changed, 11 insertions(+), 1 deletion(-) diff --git a/next-src/src/app/pages.css b/next-src/src/app/pages.css index 0b8eba13..854d6a16 100644 --- a/next-src/src/app/pages.css +++ b/next-src/src/app/pages.css @@ -1014,3 +1014,13 @@ body:has(.arena-home) .page-hills { .eval-result-links a { font-size: 13px; padding: 10px 16px; border-radius: 10px; text-decoration: none; } .eval-result-links > span { color: var(--text-muted); font-size: 12px; } @media (max-width: 600px) { .eval-result-links { padding: 0 16px 16px; gap: 8px; } .eval-result-links > span { flex-basis: 100%; } } + +.eval-success-section { margin: 24px 0 36px; } +.eval-success-scroll { overflow-x: auto; border: 1px solid var(--border); border-radius: 16px; } +.eval-success-table { width: 100%; border-collapse: collapse; text-align: left; font-size: 14px; } +.eval-success-table th, .eval-success-table td { padding: 16px 20px; border-bottom: 1px solid var(--border); vertical-align: middle; } +.eval-success-table thead th { color: var(--text-muted); font-size: 12px; font-weight: 600; white-space: nowrap; } +.eval-success-table tbody th { font-weight: 600; min-width: 150px; overflow-wrap: anywhere; } +.eval-success-table tbody tr:last-child > * { border-bottom: 0; } +.eval-success-actions { display: flex; gap: 14px; flex-wrap: wrap; } +.eval-success-actions a { white-space: nowrap; text-underline-offset: 4px; } diff --git a/next-src/src/components/EvaluationRunner.tsx b/next-src/src/components/EvaluationRunner.tsx index 5db36ea6..596ce54a 100644 --- a/next-src/src/components/EvaluationRunner.tsx +++ b/next-src/src/components/EvaluationRunner.tsx @@ -385,7 +385,7 @@ export function EvaluationRunner() {
{blockers.length>0 &&
Before you can run
}{error &&

{error}

}Runs continue in the background until complete or cancelled.{limits.max_runtime_seconds!==null && ` Admin runtime limit: ${Math.round(limits.max_runtime_seconds/60)} minutes.`}{!ownKeys && limits.require_credits && " Runs also stop when their reserved execution credits are used; unused time is returned."}
} - {tab === 'runs' &&

Your evaluations

{!runsLoaded && !jobs.length && !runsError &&

Loading your evaluations…

}{runsError &&

{runsError}

}{runsLoaded && !jobs.length && !runsError &&

No evaluations yet. Submitted runs will appear here.

}{jobs.map(job => { const open = openRuns[job.id] ?? isActive(job.state); const body = `run-body-${job.id}`; return
+ {tab === 'runs' &&

Your evaluations

{!runsLoaded && !jobs.length && !runsError &&

Loading your evaluations…

}{runsError &&

{runsError}

}{runsLoaded && !jobs.length && !runsError &&

No evaluations yet. Submitted runs will appear here.

}{jobs.some(job => job.state === 'succeeded') &&

Successful runs

{jobs.filter(job => job.state === 'succeeded').map(job => )}
RunConstitutionModelsScenariosResults
{job.name}{job.constitution}{job.models_count}{job.scenario_count}
}{jobs.length > 0 &&

All runs

}{jobs.map(job => { const open = openRuns[job.id] ?? isActive(job.state); const body = `run-body-${job.id}`; return
{/* Active runs start open; finished ones stay collapsed until clicked */} {job.state === 'succeeded' &&
View results →TranscriptsRankings and individual judgments
}