Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 14 additions & 1 deletion backend/app/crud/assessment/batch.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,13 +25,15 @@
AssessmentSubmission,
)
from app.models.batch_job import BatchJob, BatchJobType
from app.models.config.assessment_blob import VIDEO_COLUMN_TYPE
from app.models.llm.constants import DEFAULT_ASSESSMENT_BATCH_MAX_TOKENS
from app.models.llm.request import ConfigBlob
from app.services.assessment.utils.attachments import (
attachment_type_for_row,
build_anthropic_attachment_parts,
build_gemini_attachment_parts,
resolve_attachment_values,
resolve_item_type,
rewrite_gcs_attachment_urls,
)
from app.services.assessment.validators import (
Expand Down Expand Up @@ -189,13 +191,21 @@ def build_google_jsonl(
parts.append({"text": text_prompt})

# Attachments (Gemini uses file_data for inline content)
video_part_config = google_params.get("video_part_config")
has_video = False
for att in attachments:
cell_value = row.get(att.column, "")
type_override = attachment_type_for_row(att, row)
if cell_value.strip() and (
resolve_item_type(att.type, type_override) == VIDEO_COLUMN_TYPE
):
has_video = True
parts.extend(
build_gemini_attachment_parts(
cell_value,
att,
type_override=attachment_type_for_row(att, row),
type_override=type_override,
video_part_config=video_part_config,
)
)

Expand Down Expand Up @@ -227,6 +237,9 @@ def build_google_jsonl(
if output_schema:
generation_config["responseMimeType"] = "application/json"
generation_config["responseSchema"] = output_schema
media_resolution = google_params.get("media_resolution")
if media_resolution and has_video:
generation_config["mediaResolution"] = media_resolution
if generation_config:
request["generationConfig"] = generation_config

Expand Down
12 changes: 8 additions & 4 deletions backend/app/models/assessment/assessment.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@
from sqlmodel import Relationship, SQLModel

from app.core.util import now
from app.models.config.assessment_blob import AttachmentColumnType

if TYPE_CHECKING:
from app.models.batch_job import BatchJob
Expand Down Expand Up @@ -83,16 +84,19 @@ class AssessmentAttachment(BaseModel):
"""External-dataset attachment column config (RUN / BATCH-by-ref)."""

column: str = Field(..., description="Dataset column holding the attachment")
type: Literal["image", "pdf", "mixed"] = Field(
type: AttachmentColumnType | Literal["mixed"] = Field(
...,
description="'image'/'pdf' fix the type; 'mixed' resolves per-row via type_column",
description=(
"'image'/'pdf'/'video' fix the type; 'mixed' resolves per-row via type_column"
),
)
format: Literal["url", "base64"] = Field(..., description="Data format")
type_column: str | None = Field(
None, description="'mixed' only: column whose value decides each row's type"
)
type_value_map: dict[str, Literal["image", "pdf"]] | None = Field(
None, description="'mixed' only: maps a type_column value to 'image' or 'pdf'"
type_value_map: dict[str, AttachmentColumnType] | None = Field(
None,
description="'mixed' only: maps a type_column value to 'image', 'pdf' or 'video'",
)

@model_validator(mode="after")
Expand Down
17 changes: 16 additions & 1 deletion backend/app/models/config/assessment_blob.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,15 @@
# object-typed dict. Provider strict-mode normalisation is a run-mode concern.
JSON_SCHEMA_OBJECT_TYPE = "object"

IMAGE_COLUMN_TYPE = "image"
PDF_COLUMN_TYPE = "pdf"
VIDEO_COLUMN_TYPE = "video"
# Input-column types carrying a media reference rather than prompt text.
ATTACHMENT_COLUMN_TYPES = (IMAGE_COLUMN_TYPE, PDF_COLUMN_TYPE, VIDEO_COLUMN_TYPE)

AttachmentColumnType = Literal["image", "pdf", "video"]
ColumnType = Literal["text"] | AttachmentColumnType

# {column} placeholders in a submission template; the capture group is the column name.
PLACEHOLDER_RE = re.compile(r"\{(\w+)\}")

Expand All @@ -36,10 +45,16 @@ class InputColumn(SQLModel):

model_config = {"extra": "forbid"}

type: Literal["text", "image", "pdf"]
type: ColumnType
format: Literal["url", "base64"] | None = None
strict: bool = False

@model_validator(mode="after")
def _validate_video_format(self):
if self.type == VIDEO_COLUMN_TYPE and self.format == "base64":
raise ValueError("A 'video' input column must be url-format.")
return self


class PreFilterParams(TextLLMParams):
"""Flat, mapper-ready LLM params for a pre-filter call.
Expand Down
5 changes: 3 additions & 2 deletions backend/app/services/assessment/api/batch.py
Original file line number Diff line number Diff line change
Expand Up @@ -61,6 +61,7 @@
)
from app.models.batch_job import BatchJob, BatchJobType
from app.models.config.assessment_blob import (
ATTACHMENT_COLUMN_TYPES,
AssessmentConfigBlob,
AssessmentPreFilters,
TopicRelevanceFilter,
Expand Down Expand Up @@ -206,14 +207,14 @@ def column_kinds(
) -> tuple[list[str], list[AssessmentAttachment]]:
"""Split columns into text names and attachment specs per the config's ``input_schema``.

A column typed image/pdf is an attachment (url-format only); anything else is text.
A column typed image/pdf/video is an attachment (url-format only); anything else is text.
"""
text_columns: list[str] = []
attachments: list[AssessmentAttachment] = []
for column in columns:
spec = input_columns.get(column) or {}
col_type = spec.get("type", "text")
if col_type in ("image", "pdf"):
if col_type in ATTACHMENT_COLUMN_TYPES:
if (spec.get("format") or "url") != "url":
raise ValueError(
f"BATCH attachment column '{column}' must be url-format; base64 is "
Expand Down
17 changes: 14 additions & 3 deletions backend/app/services/assessment/api/submission.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,7 +26,10 @@
BatchRunState,
derive_method,
)
from app.models.config.assessment_blob import AssessmentConfigBlob
from app.models.config.assessment_blob import (
ATTACHMENT_COLUMN_TYPES,
AssessmentConfigBlob,
)
from app.models.config.config import ConfigTag
from app.services.assessment.api import batch as batch_service
from app.services.assessment.api.submission_store import (
Expand All @@ -39,14 +42,22 @@

# Attachment cell values are provided as URLs (base64 is unsupported for batch).
_URL_PREFIXES = ("http://", "https://", "gs://")
_ATTACHMENT_TYPES = ("image", "pdf")
_EXTENSION_TYPES = {
".pdf": "pdf",
".png": "image",
".jpg": "image",
".jpeg": "image",
".gif": "image",
".webp": "image",
".mp4": "video",
".mov": "video",
".mpeg": "video",
".mpg": "video",
".avi": "video",
".wmv": "video",
".flv": "video",
".webm": "video",
".3gp": "video",
}


Expand Down Expand Up @@ -93,7 +104,7 @@ def _validate_rows_against_schema(
for column, spec in input_schema.items():
column_type = (spec or {}).get("type")
value = (row.get(column) or "").strip()
if value and column_type in _ATTACHMENT_TYPES:
if value and column_type in ATTACHMENT_COLUMN_TYPES:
if not value.startswith(_URL_PREFIXES):
raise HTTPException(
status_code=422,
Expand Down
88 changes: 70 additions & 18 deletions backend/app/services/assessment/utils/attachments.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,13 +2,19 @@

import logging
import re
from typing import Any, cast
from typing import cast
from urllib.parse import urlparse

from pydantic import JsonValue
from sqlmodel import Session

from app.core.config import settings
from app.models.assessment import AssessmentAttachment
from app.models.config.assessment_blob import (
ATTACHMENT_COLUMN_TYPES,
IMAGE_COLUMN_TYPE,
VIDEO_COLUMN_TYPE,
)
from app.models.llm.constants import KaapiProvider
from app.services.buckets.attachments import is_gcs_uri, resolve_attachments

Expand All @@ -27,6 +33,24 @@
".heif": "image/heif",
}

_VIDEO_MIME_BY_EXT = {
".mp4": "video/mp4",
".mov": "video/mov",
".mpeg": "video/mpeg",
".mpg": "video/mpg",
".avi": "video/avi",
".wmv": "video/wmv",
".flv": "video/x-flv",
".webm": "video/webm",
".3gp": "video/3gpp",
}

_PDF_MIME = "application/pdf"
_DEFAULT_IMAGE_MIME = "image/png"
_DEFAULT_VIDEO_MIME = "video/mp4"

YOUTUBE_HOSTS = ("youtube.com", "www.youtube.com", "youtu.be", "m.youtube.com")


def split_attachment_urls(value: str) -> list[str]:
"""Split comma/newline separated attachment URLs from a single dataset cell."""
Expand Down Expand Up @@ -125,15 +149,23 @@ def _guess_image_mime_from_url(url: str) -> str | None:
return None


def _resolve_video_mime_from_url(url: str) -> str | None:
path = urlparse(url).path or ""
for ext, mime in _VIDEO_MIME_BY_EXT.items():
if path.lower().endswith(ext):
return mime
return None


def resolve_item_type(declared: str, type_override: str | None = None) -> str | None:
"""Resolve an attachment item as 'image' or 'pdf' from the user-declared type.
"""Resolve an attachment item as 'image', 'pdf' or 'video' from the declared type.

A per-row ``type_override`` (for 'mixed' columns) wins, else the column's declared
``type``. Returns None when the type stays unresolved (e.g. a 'mixed' row whose
value didn't map to a concrete type) so callers can skip rather than guess.
"""
item_type = type_override or declared
return item_type if item_type in ("image", "pdf") else None
return item_type if item_type in ATTACHMENT_COLUMN_TYPES else None


def _normalize_type_value(value: str) -> str:
Expand All @@ -153,7 +185,7 @@ def attachment_type_for_row(
) -> str | None:
"""For a 'mixed' column, resolve this row's type from type_column + type_value_map.

Returns 'image'/'pdf', or None to let normal detection (extension/declared) decide.
Returns a concrete attachment type, or None to let normal detection decide.
"""
type_column = getattr(att, "type_column", None)
type_value_map = getattr(att, "type_value_map", None)
Expand All @@ -162,7 +194,7 @@ def attachment_type_for_row(

normalized_map: dict[str, str] = {}
for raw_values, mapped_type in type_value_map.items():
if mapped_type not in ("image", "pdf"):
if mapped_type not in ATTACHMENT_COLUMN_TYPES:
continue
for value in _split_type_values(raw_values):
normalized_map[value] = mapped_type
Expand All @@ -181,8 +213,8 @@ def resolve_attachment_values(
value: str,
att: AssessmentAttachment,
type_override: str | None = None,
) -> list[dict[str, Any]]:
"""Convert one dataset cell into one or more OpenAI-style input objects (by URL)."""
) -> list[dict[str, JsonValue]]:
"""Resolve one dataset cell into OpenAI-supported content params (by URL)."""
value = value.strip()
if not value:
return []
Expand All @@ -194,10 +226,13 @@ def resolve_attachment_values(
att.column,
)
return []
resolved: list[dict[str, Any]] = []
# Openai doesn't support video attachments
if item_type == VIDEO_COLUMN_TYPE:
return []
resolved: list[dict[str, JsonValue]] = []
for item_value in split_attachment_urls(value):
url = to_direct_attachment_url(item_value, item_type)
if item_type == "image":
if item_type == IMAGE_COLUMN_TYPE:
resolved.append({"type": "input_image", "image_url": url})
else:
resolved.append({"type": "input_file", "file_url": url})
Expand All @@ -208,7 +243,7 @@ def build_anthropic_attachment_parts(
value: str,
att: AssessmentAttachment,
type_override: str | None = None,
) -> list[dict[str, Any]]:
) -> list[dict[str, JsonValue]]:
"""Convert one dataset cell into one or more Anthropic content blocks (by URL)."""
value = value.strip()
if not value:
Expand All @@ -221,25 +256,40 @@ def build_anthropic_attachment_parts(
att.column,
)
return []
blocks: list[dict[str, Any]] = []
# Anthropic doesn't support video attachments
if item_type == VIDEO_COLUMN_TYPE:
return []
blocks: list[dict[str, JsonValue]] = []
for item_value in split_attachment_urls(value):
url = to_direct_attachment_url(item_value, item_type)
if item_type == "image":
if item_type == IMAGE_COLUMN_TYPE:
blocks.append({"type": "image", "source": {"type": "url", "url": url}})
else:
blocks.append({"type": "document", "source": {"type": "url", "url": url}})
return blocks


def build_gemini_video_part(
url: str, video_part_config: dict[str, JsonValue] | None = None
) -> dict[str, JsonValue]:
"""One Gemini ``fileData`` video part; a YouTube url must carry no mimeType."""
file_data: dict[str, JsonValue] = {"fileUri": url}
if (urlparse(url).hostname or "").lower() not in YOUTUBE_HOSTS:
file_data["mimeType"] = _resolve_video_mime_from_url(url) or _DEFAULT_VIDEO_MIME
return {"fileData": file_data, **(video_part_config or {})}


def build_gemini_attachment_parts(
value: str,
att: AssessmentAttachment,
type_override: str | None = None,
) -> list[dict[str, Any]]:
video_part_config: dict[str, JsonValue] | None = None,
) -> list[dict[str, JsonValue]]:
"""Convert one dataset cell into one or more Gemini content parts (by URL).

Mirrors the per-item type routing used for the L2 batch so the same
image/pdf handling applies to prefilter (topic relevance) calls.
image/pdf/video handling applies to prefilter (topic relevance) calls.
``video_part_config`` comes from the Google mapper, already wire-shaped.
"""
value = value.strip()
if not value:
Expand All @@ -252,12 +302,14 @@ def build_gemini_attachment_parts(
att.column,
)
return []
parts: list[dict[str, Any]] = []
parts: list[dict[str, JsonValue]] = []
for item_value in split_attachment_urls(value):
url = to_direct_attachment_url(item_value, item_type)
if item_type == "image":
mime_type = _guess_image_mime_from_url(url) or "image/png"
if item_type == IMAGE_COLUMN_TYPE:
mime_type = _guess_image_mime_from_url(url) or _DEFAULT_IMAGE_MIME
parts.append({"fileData": {"mimeType": mime_type, "fileUri": url}})
elif item_type == VIDEO_COLUMN_TYPE:
parts.append(build_gemini_video_part(url, video_part_config))
else:
parts.append({"fileData": {"mimeType": "application/pdf", "fileUri": url}})
parts.append({"fileData": {"mimeType": _PDF_MIME, "fileUri": url}})
return parts
7 changes: 7 additions & 0 deletions backend/app/services/llm/mappers.py
Original file line number Diff line number Diff line change
Expand Up @@ -272,6 +272,9 @@ def map_kaapi_to_google_params(
- thinking_level → thinking_config.thinking_level (text only)
- output_schema → output_schema, converted to Gemini's shape (text only)
- knowledge_base_ids → FileSearch tool store names (text only)
- video_part_config → Gemini videoMetadata defaults carried on every video
content part (text only)
- media_resolution → generationConfig.mediaResolution default (text only)

Returns:
Tuple of:
Expand Down Expand Up @@ -334,6 +337,10 @@ def map_kaapi_to_google_params(
output_schema
)

# Gemini 2.5 rejects a Part-level mediaResolution; generationConfig works on 2.5 and 3.
google_params["video_part_config"] = {"videoMetadata": {"fps": 1.0}}
google_params["media_resolution"] = "MEDIA_RESOLUTION_LOW"

elif completion_type == CompletionType.TTS:
# TTS mode - voice, language, response_format
# Apply smart defaults for voice and response_format (following ElevenLabs pattern)
Expand Down
Loading
Loading