From 7153abf8a64ad30677108acea70a5c24288db046 Mon Sep 17 00:00:00 2001 From: Prashant Vasudevan <71649489+vprashrex@users.noreply.github.com> Date: Wed, 23 Sep 2026 11:53:12 +0530 Subject: [PATCH 1/5] feat(assessment): Support video attachments in batch processing and validation --- backend/app/crud/assessment/batch.py | 2 + backend/app/models/assessment/assessment.py | 11 ++- backend/app/models/config/assessment_blob.py | 14 +++- backend/app/services/assessment/api/batch.py | 5 +- .../app/services/assessment/api/submission.py | 17 ++++- .../services/assessment/utils/attachments.py | 73 ++++++++++++++++--- backend/app/services/llm/mappers.py | 8 ++ docs/wiki/modules/assessment.md | 2 +- 8 files changed, 110 insertions(+), 22 deletions(-) diff --git a/backend/app/crud/assessment/batch.py b/backend/app/crud/assessment/batch.py index 7f37d0210..bb46bbe2f 100644 --- a/backend/app/crud/assessment/batch.py +++ b/backend/app/crud/assessment/batch.py @@ -189,6 +189,7 @@ def build_google_jsonl( parts.append({"text": text_prompt}) # Attachments (Gemini uses file_data for inline content) + video_part_config = google_params.get("video_part_config") for att in attachments: cell_value = row.get(att.column, "") parts.extend( @@ -196,6 +197,7 @@ def build_google_jsonl( cell_value, att, type_override=attachment_type_for_row(att, row), + video_part_config=video_part_config, ) ) diff --git a/backend/app/models/assessment/assessment.py b/backend/app/models/assessment/assessment.py index ca1277009..7883de9ec 100644 --- a/backend/app/models/assessment/assessment.py +++ b/backend/app/models/assessment/assessment.py @@ -83,16 +83,19 @@ class AssessmentAttachment(BaseModel): """External-dataset attachment column config (RUN / BATCH-by-ref).""" column: str = Field(..., description="Dataset column holding the attachment") - type: Literal["image", "pdf", "mixed"] = Field( + type: Literal["image", "pdf", "video", "mixed"] = Field( ..., - description="'image'/'pdf' fix the type; 'mixed' resolves per-row via type_column", + description=( + "'image'/'pdf'/'video' fix the type; 'mixed' resolves per-row via type_column" + ), ) format: Literal["url", "base64"] = Field(..., description="Data format") type_column: str | None = Field( None, description="'mixed' only: column whose value decides each row's type" ) - type_value_map: dict[str, Literal["image", "pdf"]] | None = Field( - None, description="'mixed' only: maps a type_column value to 'image' or 'pdf'" + type_value_map: dict[str, Literal["image", "pdf", "video"]] | None = Field( + None, + description="'mixed' only: maps a type_column value to 'image', 'pdf' or 'video'", ) @model_validator(mode="after") diff --git a/backend/app/models/config/assessment_blob.py b/backend/app/models/config/assessment_blob.py index f30f9f6ab..4703021c2 100644 --- a/backend/app/models/config/assessment_blob.py +++ b/backend/app/models/config/assessment_blob.py @@ -18,6 +18,12 @@ # object-typed dict. Provider strict-mode normalisation is a run-mode concern. JSON_SCHEMA_OBJECT_TYPE = "object" +IMAGE_COLUMN_TYPE = "image" +PDF_COLUMN_TYPE = "pdf" +VIDEO_COLUMN_TYPE = "video" +# Input-column types carrying a media reference rather than prompt text. +ATTACHMENT_COLUMN_TYPES = (IMAGE_COLUMN_TYPE, PDF_COLUMN_TYPE, VIDEO_COLUMN_TYPE) + # {column} placeholders in a submission template; the capture group is the column name. PLACEHOLDER_RE = re.compile(r"\{(\w+)\}") @@ -36,10 +42,16 @@ class InputColumn(SQLModel): model_config = {"extra": "forbid"} - type: Literal["text", "image", "pdf"] + type: Literal["text", "image", "pdf", "video"] format: Literal["url", "base64"] | None = None strict: bool = False + @model_validator(mode="after") + def _validate_video_format(self): + if self.type == VIDEO_COLUMN_TYPE and self.format == "base64": + raise ValueError("A 'video' input column must be url-format.") + return self + class PreFilterParams(TextLLMParams): """Flat, mapper-ready LLM params for a pre-filter call. diff --git a/backend/app/services/assessment/api/batch.py b/backend/app/services/assessment/api/batch.py index ec8dfdba9..3f296e552 100644 --- a/backend/app/services/assessment/api/batch.py +++ b/backend/app/services/assessment/api/batch.py @@ -61,6 +61,7 @@ ) from app.models.batch_job import BatchJob, BatchJobType from app.models.config.assessment_blob import ( + ATTACHMENT_COLUMN_TYPES, AssessmentConfigBlob, AssessmentPreFilters, TopicRelevanceFilter, @@ -206,14 +207,14 @@ def column_kinds( ) -> tuple[list[str], list[AssessmentAttachment]]: """Split columns into text names and attachment specs per the config's ``input_schema``. - A column typed image/pdf is an attachment (url-format only); anything else is text. + A column typed image/pdf/video is an attachment (url-format only); anything else is text. """ text_columns: list[str] = [] attachments: list[AssessmentAttachment] = [] for column in columns: spec = input_columns.get(column) or {} col_type = spec.get("type", "text") - if col_type in ("image", "pdf"): + if col_type in ATTACHMENT_COLUMN_TYPES: if (spec.get("format") or "url") != "url": raise ValueError( f"BATCH attachment column '{column}' must be url-format; base64 is " diff --git a/backend/app/services/assessment/api/submission.py b/backend/app/services/assessment/api/submission.py index 848466fae..c8a3a934a 100644 --- a/backend/app/services/assessment/api/submission.py +++ b/backend/app/services/assessment/api/submission.py @@ -26,7 +26,10 @@ BatchRunState, derive_method, ) -from app.models.config.assessment_blob import AssessmentConfigBlob +from app.models.config.assessment_blob import ( + ATTACHMENT_COLUMN_TYPES, + AssessmentConfigBlob, +) from app.models.config.config import ConfigTag from app.services.assessment.api import batch as batch_service from app.services.assessment.api.submission_store import ( @@ -39,7 +42,6 @@ # Attachment cell values are provided as URLs (base64 is unsupported for batch). _URL_PREFIXES = ("http://", "https://", "gs://") -_ATTACHMENT_TYPES = ("image", "pdf") _EXTENSION_TYPES = { ".pdf": "pdf", ".png": "image", @@ -47,6 +49,15 @@ ".jpeg": "image", ".gif": "image", ".webp": "image", + ".mp4": "video", + ".mov": "video", + ".mpeg": "video", + ".mpg": "video", + ".avi": "video", + ".wmv": "video", + ".flv": "video", + ".webm": "video", + ".3gp": "video", } @@ -93,7 +104,7 @@ def _validate_rows_against_schema( for column, spec in input_schema.items(): column_type = (spec or {}).get("type") value = (row.get(column) or "").strip() - if value and column_type in _ATTACHMENT_TYPES: + if value and column_type in ATTACHMENT_COLUMN_TYPES: if not value.startswith(_URL_PREFIXES): raise HTTPException( status_code=422, diff --git a/backend/app/services/assessment/utils/attachments.py b/backend/app/services/assessment/utils/attachments.py index d4c41af14..c02daec39 100644 --- a/backend/app/services/assessment/utils/attachments.py +++ b/backend/app/services/assessment/utils/attachments.py @@ -9,6 +9,11 @@ from app.core.config import settings from app.models.assessment import AssessmentAttachment +from app.models.config.assessment_blob import ( + ATTACHMENT_COLUMN_TYPES, + IMAGE_COLUMN_TYPE, + VIDEO_COLUMN_TYPE, +) from app.models.llm.constants import KaapiProvider from app.services.buckets.attachments import is_gcs_uri, resolve_attachments @@ -27,6 +32,24 @@ ".heif": "image/heif", } +_VIDEO_MIME_BY_EXT = { + ".mp4": "video/mp4", + ".mov": "video/mov", + ".mpeg": "video/mpeg", + ".mpg": "video/mpg", + ".avi": "video/avi", + ".wmv": "video/wmv", + ".flv": "video/x-flv", + ".webm": "video/webm", + ".3gp": "video/3gpp", +} + +_PDF_MIME = "application/pdf" +_DEFAULT_IMAGE_MIME = "image/png" +_DEFAULT_VIDEO_MIME = "video/mp4" + +YOUTUBE_HOSTS = ("youtube.com", "www.youtube.com", "youtu.be", "m.youtube.com") + def split_attachment_urls(value: str) -> list[str]: """Split comma/newline separated attachment URLs from a single dataset cell.""" @@ -125,15 +148,23 @@ def _guess_image_mime_from_url(url: str) -> str | None: return None +def _resolve_video_mime_from_url(url: str) -> str | None: + path = urlparse(url).path or "" + for ext, mime in _VIDEO_MIME_BY_EXT.items(): + if path.lower().endswith(ext): + return mime + return None + + def resolve_item_type(declared: str, type_override: str | None = None) -> str | None: - """Resolve an attachment item as 'image' or 'pdf' from the user-declared type. + """Resolve an attachment item as 'image', 'pdf' or 'video' from the declared type. A per-row ``type_override`` (for 'mixed' columns) wins, else the column's declared ``type``. Returns None when the type stays unresolved (e.g. a 'mixed' row whose value didn't map to a concrete type) so callers can skip rather than guess. """ item_type = type_override or declared - return item_type if item_type in ("image", "pdf") else None + return item_type if item_type in ATTACHMENT_COLUMN_TYPES else None def _normalize_type_value(value: str) -> str: @@ -153,7 +184,7 @@ def attachment_type_for_row( ) -> str | None: """For a 'mixed' column, resolve this row's type from type_column + type_value_map. - Returns 'image'/'pdf', or None to let normal detection (extension/declared) decide. + Returns a concrete attachment type, or None to let normal detection decide. """ type_column = getattr(att, "type_column", None) type_value_map = getattr(att, "type_value_map", None) @@ -162,7 +193,7 @@ def attachment_type_for_row( normalized_map: dict[str, str] = {} for raw_values, mapped_type in type_value_map.items(): - if mapped_type not in ("image", "pdf"): + if mapped_type not in ATTACHMENT_COLUMN_TYPES: continue for value in _split_type_values(raw_values): normalized_map[value] = mapped_type @@ -182,7 +213,7 @@ def resolve_attachment_values( att: AssessmentAttachment, type_override: str | None = None, ) -> list[dict[str, Any]]: - """Convert one dataset cell into one or more OpenAI-style input objects (by URL).""" + """Resolve one dataset cell into OpenAI-supported content params (by URL).""" value = value.strip() if not value: return [] @@ -194,10 +225,13 @@ def resolve_attachment_values( att.column, ) return [] + # Openai doesn't support video attachments + if item_type == VIDEO_COLUMN_TYPE: + return [] resolved: list[dict[str, Any]] = [] for item_value in split_attachment_urls(value): url = to_direct_attachment_url(item_value, item_type) - if item_type == "image": + if item_type == IMAGE_COLUMN_TYPE: resolved.append({"type": "input_image", "image_url": url}) else: resolved.append({"type": "input_file", "file_url": url}) @@ -221,25 +255,40 @@ def build_anthropic_attachment_parts( att.column, ) return [] + # Anthropic doesn't support video attachments + if item_type == VIDEO_COLUMN_TYPE: + return [] blocks: list[dict[str, Any]] = [] for item_value in split_attachment_urls(value): url = to_direct_attachment_url(item_value, item_type) - if item_type == "image": + if item_type == IMAGE_COLUMN_TYPE: blocks.append({"type": "image", "source": {"type": "url", "url": url}}) else: blocks.append({"type": "document", "source": {"type": "url", "url": url}}) return blocks +def build_gemini_video_part( + url: str, video_part_config: dict[str, Any] | None = None +) -> dict[str, Any]: + """One Gemini ``fileData`` video part; a YouTube url must carry no mimeType.""" + file_data: dict[str, Any] = {"fileUri": url} + if (urlparse(url).hostname or "").lower() not in YOUTUBE_HOSTS: + file_data["mimeType"] = _resolve_video_mime_from_url(url) or _DEFAULT_VIDEO_MIME + return {"fileData": file_data, **(video_part_config or {})} + + def build_gemini_attachment_parts( value: str, att: AssessmentAttachment, type_override: str | None = None, + video_part_config: dict[str, Any] | None = None, ) -> list[dict[str, Any]]: """Convert one dataset cell into one or more Gemini content parts (by URL). Mirrors the per-item type routing used for the L2 batch so the same - image/pdf handling applies to prefilter (topic relevance) calls. + image/pdf/video handling applies to prefilter (topic relevance) calls. + ``video_part_config`` comes from the Google mapper, already wire-shaped. """ value = value.strip() if not value: @@ -255,9 +304,11 @@ def build_gemini_attachment_parts( parts: list[dict[str, Any]] = [] for item_value in split_attachment_urls(value): url = to_direct_attachment_url(item_value, item_type) - if item_type == "image": - mime_type = _guess_image_mime_from_url(url) or "image/png" + if item_type == IMAGE_COLUMN_TYPE: + mime_type = _guess_image_mime_from_url(url) or _DEFAULT_IMAGE_MIME parts.append({"fileData": {"mimeType": mime_type, "fileUri": url}}) + elif item_type == VIDEO_COLUMN_TYPE: + parts.append(build_gemini_video_part(url, video_part_config)) else: - parts.append({"fileData": {"mimeType": "application/pdf", "fileUri": url}}) + parts.append({"fileData": {"mimeType": _PDF_MIME, "fileUri": url}}) return parts diff --git a/backend/app/services/llm/mappers.py b/backend/app/services/llm/mappers.py index 88c26cade..75eaf7b17 100644 --- a/backend/app/services/llm/mappers.py +++ b/backend/app/services/llm/mappers.py @@ -272,6 +272,8 @@ def map_kaapi_to_google_params( - thinking_level → thinking_config.thinking_level (text only) - output_schema → output_schema, converted to Gemini's shape (text only) - knowledge_base_ids → FileSearch tool store names (text only) + - video_part_config → Gemini videoMetadata/mediaResolution defaults carried + on every video content part (text only) Returns: Tuple of: @@ -334,6 +336,12 @@ def map_kaapi_to_google_params( output_schema ) + # NOTE: Google Gemini Video Config (using default values for fps and mediaResolution) + google_params["video_part_config"] = { + "videoMetadata": {"fps": 1.0}, + "mediaResolution": {"level": "MEDIA_RESOLUTION_LOW"}, + } + elif completion_type == CompletionType.TTS: # TTS mode - voice, language, response_format # Apply smart defaults for voice and response_format (following ElevenLabs pattern) diff --git a/docs/wiki/modules/assessment.md b/docs/wiki/modules/assessment.md index bb09c0e47..b06db5b1e 100644 --- a/docs/wiki/modules/assessment.md +++ b/docs/wiki/modules/assessment.md @@ -17,7 +17,7 @@ All paths relative to `backend/app/`. | `assessment_run` (AssessmentRun; child — one config execution, BATCH/RUN; FK → assessment, config, batch_job) | `models/assessment/assessment.py` | | `assessment_submission` (AssessmentSubmission; an uploaded CSV/XLSX a run can read its rows from; FK → org, project) | `models/assessment/submission.py` | -Config version (tag=ASSESSMENT, `models/config/assessment_blob.py`) owns system / pre-filters / params / schemas: `input_schema` is a **top-level** field on the blob (sibling of `pre_filters`/`assessment`, **mandatory, non-empty** per-column spec `{type, format}` for the BATCH `data` rows — `type` is required per column; `strict: true` means the column must be present and non-blank in every submission row; it is **off by default** because the console declares every sheet column (the backend rejects undeclared ones) and real sheets have blanks — a blank on a non-strict column passes and `build_rows` fills omitted keys with `""` so the placeholder still resolves; note the config store persists the validated blob *with defaults*, so a saved version pins the default in force at save time; `InputColumn` is `extra=forbid`, so a misspelt flag fails at config save; attachment columns are url-format only) — it describes the shared input rows once, so both the pre-filter and assessment consumers read the same schema. `submission` (the per-row prompt template with `{column}` placeholders — **mandatory** on `assessment.params`, optional on each pre-filter's `params`; every `{placeholder}` is validated at config save against the top-level `input_schema` keys, and an unknown placeholder rejects the save). `json_output_schema` stays in `assessment.params` (assessment-specific, object-typed structured-output schema, omit for free text). The only API-client pre-filter is `topic_relevance` (duplicate_detection was removed from the API-client pipeline; it survives only in the legacy RUN pipeline); it carries its own `provider` (default `openai`) + `params` (TextLLMParams: model, temperature, ...) and runs its own llm call; its criteria live in `params.instructions` (a **mandatory** field, same shape as the assessment call — pre-filters no longer have a top-level `prompt`/`content`) and it may carry its own `params.submission` template. The prompt template lives on the config, not the request. Strict input types `ResponseInput` (RESPONSE, `{attachments}`) / `BatchInput` (BATCH) no longer carry `query`; they discriminate structurally (a `data` or `submission_doc_id` key ⇒ BATCH, else RESPONSE) with `extra=forbid` keeping them disjoint and rejecting a stray `query` — no `mode` tag (`models/assessment/assessment_api.py`). `BatchInput` takes the rows **either** inline as `data` (a list of submission rows, each a flat column→string map, an attachment column's value being a url string) **or** by reference as `submission_doc_id` (an `assessment_submission` id); a `model_validator` requires exactly one, so both-or-neither is a 422. A `submission_doc_id` is resolved at submit: its rows are read, projected onto the schema's columns (a sheet column `input_schema` does not name is dropped, so the schema is the selection and the console need only declare the `@`-referenced columns), validated against `input_schema` like inline ones, and copied into that run's own `submission.jsonl`, so downstream stays one code path and the run keeps an immutable snapshot if the submission file is later replaced. `assessment.submission_id` records which submission it came from, so the provenance survives the copy and `delete_submission` refuses while a BATCH run still points at it. That column is set by RUN and by a `submission_doc_id` BATCH alike; it stays NULL only when BATCH sent rows inline. Legacy RUN runtime lives in `assessment_run.execution` (`RunExecution`). +Config version (tag=ASSESSMENT, `models/config/assessment_blob.py`) owns system / pre-filters / params / schemas: `input_schema` is a **top-level** field on the blob (sibling of `pre_filters`/`assessment`, **mandatory, non-empty** per-column spec `{type, format}` for the BATCH `data` rows — `type` is required per column; `strict: true` means the column must be present and non-blank in every submission row; it is **off by default** because the console declares every sheet column (the backend rejects undeclared ones) and real sheets have blanks — a blank on a non-strict column passes and `build_rows` fills omitted keys with `""` so the placeholder still resolves; note the config store persists the validated blob *with defaults*, so a saved version pins the default in force at save time; `InputColumn` is `extra=forbid`, so a misspelt flag fails at config save; attachment columns (`image`/`pdf`/`video`) are url-format only, and `video` is Google-only — the OpenAI and Anthropic part builders log and skip a video cell) — it describes the shared input rows once, so both the pre-filter and assessment consumers read the same schema. `submission` (the per-row prompt template with `{column}` placeholders — **mandatory** on `assessment.params`, optional on each pre-filter's `params`; every `{placeholder}` is validated at config save against the top-level `input_schema` keys, and an unknown placeholder rejects the save). `json_output_schema` stays in `assessment.params` (assessment-specific, object-typed structured-output schema, omit for free text). The only API-client pre-filter is `topic_relevance` (duplicate_detection was removed from the API-client pipeline; it survives only in the legacy RUN pipeline); it carries its own `provider` (default `openai`) + `params` (TextLLMParams: model, temperature, ...) and runs its own llm call; its criteria live in `params.instructions` (a **mandatory** field, same shape as the assessment call — pre-filters no longer have a top-level `prompt`/`content`) and it may carry its own `params.submission` template. The prompt template lives on the config, not the request. Strict input types `ResponseInput` (RESPONSE, `{attachments}`) / `BatchInput` (BATCH) no longer carry `query`; they discriminate structurally (a `data` or `submission_doc_id` key ⇒ BATCH, else RESPONSE) with `extra=forbid` keeping them disjoint and rejecting a stray `query` — no `mode` tag (`models/assessment/assessment_api.py`). `BatchInput` takes the rows **either** inline as `data` (a list of submission rows, each a flat column→string map, an attachment column's value being a url string) **or** by reference as `submission_doc_id` (an `assessment_submission` id); a `model_validator` requires exactly one, so both-or-neither is a 422. A `submission_doc_id` is resolved at submit: its rows are read, projected onto the schema's columns (a sheet column `input_schema` does not name is dropped, so the schema is the selection and the console need only declare the `@`-referenced columns), validated against `input_schema` like inline ones, and copied into that run's own `submission.jsonl`, so downstream stays one code path and the run keeps an immutable snapshot if the submission file is later replaced. `assessment.submission_id` records which submission it came from, so the provenance survives the copy and `delete_submission` refuses while a BATCH run still points at it. That column is set by RUN and by a `submission_doc_id` BATCH alike; it stays NULL only when BATCH sent rows inline. Legacy RUN runtime lives in `assessment_run.execution` (`RunExecution`). `assessment.id` is a **UUID** (like config/job/llm_call). Per-item result = `AssessmentResult {output: {assessment, pre_filter}, error}` (no `metadata` — the provider/model/usage block was removed from the API-client output) where `output.assessment` = the LLM output parsed to an object when the config has a `json_output_schema`, else string (null for gated/failed rows), and `output.pre_filter` holds the `{topic_relevance}` verdict (`{verdict, reasoning}` or null) and is itself null when no pre-filter ran. The API-client BATCH rows do **not** live in `assessment.input` (that column is now RESPONSE/RUN only, NULL for BATCH). They are uploaded to `submission.jsonl` at submit and `assessment.submission_input` holds its `s3://` url: 3-6MB of JSONB was dragged along by every full-row `SELECT` of the assessment. `services/assessment/api/submission_store.py` owns the round trip: `upload_submission_rows` at submit, and the `open_submission_rows` context manager on read, which streams the JSONL line by line and closes the body on exit. Rows are read only inside `_submit_stage`, never on a poll, and only the rows the stage's subset needs are held (the count comes from `execution.total_items`, so an empty subset never opens the file). A storage read failure raises `SubmissionUnavailableError`, which requeues the task instead of failing the execution; a corrupt line is a `ValueError` and terminal. An upload failure at submit is a 503 (an assessment without its rows is unrunnable). `build_result` reads `execution.total_items` rather than re-deriving the count from the rows, so the terminal path never fetches them. From 06d990ad33a63c2cbb384f16b589198d39651957 Mon Sep 17 00:00:00 2001 From: Prashant Vasudevan <71649489+vprashrex@users.noreply.github.com> Date: Tue, 29 Sep 2026 18:19:40 +0530 Subject: [PATCH 2/5] feat(assessment): Enhance video handling in JSONL generation and mapping --- backend/app/crud/assessment/batch.py | 13 ++++++++++++- backend/app/services/llm/mappers.py | 13 ++++++------- 2 files changed, 18 insertions(+), 8 deletions(-) diff --git a/backend/app/crud/assessment/batch.py b/backend/app/crud/assessment/batch.py index bb46bbe2f..d1014d414 100644 --- a/backend/app/crud/assessment/batch.py +++ b/backend/app/crud/assessment/batch.py @@ -25,6 +25,7 @@ AssessmentSubmission, ) from app.models.batch_job import BatchJob, BatchJobType +from app.models.config.assessment_blob import VIDEO_COLUMN_TYPE from app.models.llm.constants import DEFAULT_ASSESSMENT_BATCH_MAX_TOKENS from app.models.llm.request import ConfigBlob from app.services.assessment.utils.attachments import ( @@ -32,6 +33,7 @@ build_anthropic_attachment_parts, build_gemini_attachment_parts, resolve_attachment_values, + resolve_item_type, rewrite_gcs_attachment_urls, ) from app.services.assessment.validators import ( @@ -190,13 +192,19 @@ def build_google_jsonl( # Attachments (Gemini uses file_data for inline content) video_part_config = google_params.get("video_part_config") + has_video = False for att in attachments: cell_value = row.get(att.column, "") + type_override = attachment_type_for_row(att, row) + if cell_value.strip() and ( + resolve_item_type(att.type, type_override) == VIDEO_COLUMN_TYPE + ): + has_video = True parts.extend( build_gemini_attachment_parts( cell_value, att, - type_override=attachment_type_for_row(att, row), + type_override=type_override, video_part_config=video_part_config, ) ) @@ -229,6 +237,9 @@ def build_google_jsonl( if output_schema: generation_config["responseMimeType"] = "application/json" generation_config["responseSchema"] = output_schema + media_resolution = google_params.get("media_resolution") + if media_resolution and has_video: + generation_config["mediaResolution"] = media_resolution if generation_config: request["generationConfig"] = generation_config diff --git a/backend/app/services/llm/mappers.py b/backend/app/services/llm/mappers.py index 75eaf7b17..53fec7741 100644 --- a/backend/app/services/llm/mappers.py +++ b/backend/app/services/llm/mappers.py @@ -272,8 +272,9 @@ def map_kaapi_to_google_params( - thinking_level → thinking_config.thinking_level (text only) - output_schema → output_schema, converted to Gemini's shape (text only) - knowledge_base_ids → FileSearch tool store names (text only) - - video_part_config → Gemini videoMetadata/mediaResolution defaults carried - on every video content part (text only) + - video_part_config → Gemini videoMetadata defaults carried on every video + content part (text only) + - media_resolution → generationConfig.mediaResolution default (text only) Returns: Tuple of: @@ -336,11 +337,9 @@ def map_kaapi_to_google_params( output_schema ) - # NOTE: Google Gemini Video Config (using default values for fps and mediaResolution) - google_params["video_part_config"] = { - "videoMetadata": {"fps": 1.0}, - "mediaResolution": {"level": "MEDIA_RESOLUTION_LOW"}, - } + # Gemini 2.5 rejects a Part-level mediaResolution; generationConfig works on 2.5 and 3. + google_params["video_part_config"] = {"videoMetadata": {"fps": 1.0}} + google_params["media_resolution"] = "MEDIA_RESOLUTION_LOW" elif completion_type == CompletionType.TTS: # TTS mode - voice, language, response_format From b28befe72b21e7d40c2a2811c9cafd8608104231 Mon Sep 17 00:00:00 2001 From: Prashant Vasudevan <71649489+vprashrex@users.noreply.github.com> Date: Tue, 29 Sep 2026 18:42:28 +0530 Subject: [PATCH 3/5] feat(assessment): Add video configuration to Google params mapping test --- backend/app/tests/services/llm/test_mappers.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/backend/app/tests/services/llm/test_mappers.py b/backend/app/tests/services/llm/test_mappers.py index c69832c35..f0540813f 100644 --- a/backend/app/tests/services/llm/test_mappers.py +++ b/backend/app/tests/services/llm/test_mappers.py @@ -197,7 +197,12 @@ def test_text_completion_basic(self): kaapi_params.model_dump(exclude_none=True), completion_type="text" ) - assert result == {"model": "gemini-2.5-pro", "temperature": 0.7} + assert result == { + "model": "gemini-2.5-pro", + "temperature": 0.7, + "video_part_config": {"videoMetadata": {"fps": 1.0}}, + "media_resolution": "MEDIA_RESOLUTION_LOW", + } assert warnings == [] def test_text_completion_with_reasoning(self): From d318a8fce329d2350f37fd786e1e314f03e2366a Mon Sep 17 00:00:00 2001 From: Prashant Vasudevan <71649489+vprashrex@users.noreply.github.com> Date: Tue, 29 Sep 2026 19:26:46 +0530 Subject: [PATCH 4/5] feat(assessment): Add video handling tests for JSONL generation and validation --- backend/app/tests/assessment/test_api_crud.py | 8 ++++ backend/app/tests/assessment/test_batch.py | 41 +++++++++++++++++++ 2 files changed, 49 insertions(+) diff --git a/backend/app/tests/assessment/test_api_crud.py b/backend/app/tests/assessment/test_api_crud.py index 1ad7502e9..b49ce63c2 100644 --- a/backend/app/tests/assessment/test_api_crud.py +++ b/backend/app/tests/assessment/test_api_crud.py @@ -495,3 +495,11 @@ def test_valid_type_accepted(self) -> None: col = InputColumn.model_validate({"type": "image", "format": "url"}) assert col.type == "image" assert col.format == "url" + + def test_video_base64_rejected(self) -> None: + with pytest.raises(ValidationError): + InputColumn.model_validate({"type": "video", "format": "base64"}) + + def test_video_url_accepted(self) -> None: + col = InputColumn.model_validate({"type": "video", "format": "url"}) + assert col.type == "video" diff --git a/backend/app/tests/assessment/test_batch.py b/backend/app/tests/assessment/test_batch.py index 2662050d2..cfeb8551d 100644 --- a/backend/app/tests/assessment/test_batch.py +++ b/backend/app/tests/assessment/test_batch.py @@ -523,6 +523,23 @@ def test_build_openai_and_google_jsonl(self) -> None: "parts": [{"text": "system"}] } + def test_build_google_jsonl_with_video_sets_media_resolution(self) -> None: + rows = [{"q": "What happens?", "vid": "https://x.com/a.mp4"}] + attachments = [AssessmentAttachment(column="vid", type="video", format="url")] + + google_jsonl = build_google_jsonl( + rows=rows, + text_columns=["q"], + attachments=attachments, + prompt_template=None, + google_params={ + "video_part_config": {"videoMetadata": {"fps": 1.0}}, + "media_resolution": "MEDIA_RESOLUTION_LOW", + }, + ) + generation_config = google_jsonl[0]["request"]["generationConfig"] + assert generation_config["mediaResolution"] == "MEDIA_RESOLUTION_LOW" + def test_build_anthropic_jsonl(self) -> None: rows = [ { @@ -727,6 +744,7 @@ def test_override_forces_part_type(self) -> None: class TestAttachmentResolutionBranches: _IMG = AssessmentAttachment(column="Docs", type="image", format="url") _PDF = AssessmentAttachment(column="Docs", type="pdf", format="url") + _VIDEO = AssessmentAttachment(column="Docs", type="video", format="url") _MIXED = AssessmentAttachment( column="Docs", type="mixed", @@ -751,6 +769,29 @@ def test_gemini_image_and_pdf_parts(self) -> None: assert img["fileData"]["mimeType"] == "image/png" assert pdf["fileData"]["mimeType"] == "application/pdf" + def test_openai_and_anthropic_skip_video(self) -> None: + url = "https://x.com/a.mp4" + assert resolve_attachment_values(url, self._VIDEO) == [] + assert build_anthropic_attachment_parts(url, self._VIDEO) == [] + + def test_gemini_video_part_youtube_omits_mime_type(self) -> None: + part = build_gemini_attachment_parts( + "https://youtu.be/gKJiCGNaAwg", self._VIDEO + )[0] + assert part["fileData"] == {"fileUri": "https://youtu.be/gKJiCGNaAwg"} + + def test_gemini_video_part_direct_url_resolves_mime_and_config(self) -> None: + part = build_gemini_attachment_parts( + "https://x.com/a.mp4", + self._VIDEO, + video_part_config={"videoMetadata": {"fps": 1.0}}, + )[0] + assert part["fileData"] == { + "fileUri": "https://x.com/a.mp4", + "mimeType": "video/mp4", + } + assert part["videoMetadata"] == {"fps": 1.0} + def test_type_for_row_blank_value_returns_none(self) -> None: assert attachment_type_for_row(self._MIXED, {"DOC type": " "}) is None From cbaeff4b83646e1f872979ce8a23acde28bbbe12 Mon Sep 17 00:00:00 2001 From: Prashant Vasudevan <71649489+vprashrex@users.noreply.github.com> Date: Tue, 29 Sep 2026 19:57:46 +0530 Subject: [PATCH 5/5] feat(assessment): Refactor attachment type handling to use AttachmentColumnType --- backend/app/models/assessment/assessment.py | 5 ++-- backend/app/models/config/assessment_blob.py | 5 +++- .../services/assessment/utils/attachments.py | 23 ++++++++++--------- 3 files changed, 19 insertions(+), 14 deletions(-) diff --git a/backend/app/models/assessment/assessment.py b/backend/app/models/assessment/assessment.py index 7883de9ec..259f975ad 100644 --- a/backend/app/models/assessment/assessment.py +++ b/backend/app/models/assessment/assessment.py @@ -17,6 +17,7 @@ from sqlmodel import Relationship, SQLModel from app.core.util import now +from app.models.config.assessment_blob import AttachmentColumnType if TYPE_CHECKING: from app.models.batch_job import BatchJob @@ -83,7 +84,7 @@ class AssessmentAttachment(BaseModel): """External-dataset attachment column config (RUN / BATCH-by-ref).""" column: str = Field(..., description="Dataset column holding the attachment") - type: Literal["image", "pdf", "video", "mixed"] = Field( + type: AttachmentColumnType | Literal["mixed"] = Field( ..., description=( "'image'/'pdf'/'video' fix the type; 'mixed' resolves per-row via type_column" @@ -93,7 +94,7 @@ class AssessmentAttachment(BaseModel): type_column: str | None = Field( None, description="'mixed' only: column whose value decides each row's type" ) - type_value_map: dict[str, Literal["image", "pdf", "video"]] | None = Field( + type_value_map: dict[str, AttachmentColumnType] | None = Field( None, description="'mixed' only: maps a type_column value to 'image', 'pdf' or 'video'", ) diff --git a/backend/app/models/config/assessment_blob.py b/backend/app/models/config/assessment_blob.py index 4703021c2..8f6aed712 100644 --- a/backend/app/models/config/assessment_blob.py +++ b/backend/app/models/config/assessment_blob.py @@ -24,6 +24,9 @@ # Input-column types carrying a media reference rather than prompt text. ATTACHMENT_COLUMN_TYPES = (IMAGE_COLUMN_TYPE, PDF_COLUMN_TYPE, VIDEO_COLUMN_TYPE) +AttachmentColumnType = Literal["image", "pdf", "video"] +ColumnType = Literal["text"] | AttachmentColumnType + # {column} placeholders in a submission template; the capture group is the column name. PLACEHOLDER_RE = re.compile(r"\{(\w+)\}") @@ -42,7 +45,7 @@ class InputColumn(SQLModel): model_config = {"extra": "forbid"} - type: Literal["text", "image", "pdf", "video"] + type: ColumnType format: Literal["url", "base64"] | None = None strict: bool = False diff --git a/backend/app/services/assessment/utils/attachments.py b/backend/app/services/assessment/utils/attachments.py index c02daec39..c47724cd5 100644 --- a/backend/app/services/assessment/utils/attachments.py +++ b/backend/app/services/assessment/utils/attachments.py @@ -2,9 +2,10 @@ import logging import re -from typing import Any, cast +from typing import cast from urllib.parse import urlparse +from pydantic import JsonValue from sqlmodel import Session from app.core.config import settings @@ -212,7 +213,7 @@ def resolve_attachment_values( value: str, att: AssessmentAttachment, type_override: str | None = None, -) -> list[dict[str, Any]]: +) -> list[dict[str, JsonValue]]: """Resolve one dataset cell into OpenAI-supported content params (by URL).""" value = value.strip() if not value: @@ -228,7 +229,7 @@ def resolve_attachment_values( # Openai doesn't support video attachments if item_type == VIDEO_COLUMN_TYPE: return [] - resolved: list[dict[str, Any]] = [] + resolved: list[dict[str, JsonValue]] = [] for item_value in split_attachment_urls(value): url = to_direct_attachment_url(item_value, item_type) if item_type == IMAGE_COLUMN_TYPE: @@ -242,7 +243,7 @@ def build_anthropic_attachment_parts( value: str, att: AssessmentAttachment, type_override: str | None = None, -) -> list[dict[str, Any]]: +) -> list[dict[str, JsonValue]]: """Convert one dataset cell into one or more Anthropic content blocks (by URL).""" value = value.strip() if not value: @@ -258,7 +259,7 @@ def build_anthropic_attachment_parts( # Anthropic doesn't support video attachments if item_type == VIDEO_COLUMN_TYPE: return [] - blocks: list[dict[str, Any]] = [] + blocks: list[dict[str, JsonValue]] = [] for item_value in split_attachment_urls(value): url = to_direct_attachment_url(item_value, item_type) if item_type == IMAGE_COLUMN_TYPE: @@ -269,10 +270,10 @@ def build_anthropic_attachment_parts( def build_gemini_video_part( - url: str, video_part_config: dict[str, Any] | None = None -) -> dict[str, Any]: + url: str, video_part_config: dict[str, JsonValue] | None = None +) -> dict[str, JsonValue]: """One Gemini ``fileData`` video part; a YouTube url must carry no mimeType.""" - file_data: dict[str, Any] = {"fileUri": url} + file_data: dict[str, JsonValue] = {"fileUri": url} if (urlparse(url).hostname or "").lower() not in YOUTUBE_HOSTS: file_data["mimeType"] = _resolve_video_mime_from_url(url) or _DEFAULT_VIDEO_MIME return {"fileData": file_data, **(video_part_config or {})} @@ -282,8 +283,8 @@ def build_gemini_attachment_parts( value: str, att: AssessmentAttachment, type_override: str | None = None, - video_part_config: dict[str, Any] | None = None, -) -> list[dict[str, Any]]: + video_part_config: dict[str, JsonValue] | None = None, +) -> list[dict[str, JsonValue]]: """Convert one dataset cell into one or more Gemini content parts (by URL). Mirrors the per-item type routing used for the L2 batch so the same @@ -301,7 +302,7 @@ def build_gemini_attachment_parts( att.column, ) return [] - parts: list[dict[str, Any]] = [] + parts: list[dict[str, JsonValue]] = [] for item_value in split_attachment_urls(value): url = to_direct_attachment_url(item_value, item_type) if item_type == IMAGE_COLUMN_TYPE: