mirror of
https://github.com/andrewyng/openworker.git
synced 2026-09-03 04:49:26 +00:00
Imported from andrewyng/aisuite@1b4bbf303e (contents of its platform/ directory, hoisted to the repo root). Development history prior to this commit lives in that repository. Co-authored-by: Devika <devikaverma11@gmail.com>
100 lines
3.8 KiB
Python
100 lines
3.8 KiB
Python
"""Build OpenAI content-parts from a user message + attachments (images, PDFs, text files).
|
||
|
||
We pass messages straight to the OpenAI SDK, which accepts `content` as either a string or an
|
||
array of parts: `{"type": "text", ...}`, `{"type": "image_url", "image_url": {"url": ...}}`
|
||
(data: URLs work, and vision models read them), and `{"type": "file", "file": {"filename",
|
||
"file_data"}}` for PDFs. So image/PDF attachments are just parts appended to the user turn —
|
||
the Anthropic/Gemini providers convert them to their own block shapes.
|
||
|
||
`build_user_content` returns a plain string when there are no attachments (back-compat with the
|
||
text-only path), else the parts list.
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
from typing import Any, Optional
|
||
|
||
MAX_ATTACHMENTS = 8
|
||
MAX_IMAGE_CHARS = 12_000_000 # data-URL length cap (~8–9 MB decoded); keeps a turn sane
|
||
MAX_PDF_CHARS = 15_000_000 # data-URL length cap (~10 MB decoded, the GUI's pick limit)
|
||
MAX_TEXT_CHARS = 200_000 # per text file, inlined
|
||
|
||
|
||
def _is_data_image(url: Any) -> bool:
|
||
return isinstance(url, str) and url.startswith("data:image/") and ";base64," in url
|
||
|
||
|
||
def _is_data_pdf(url: Any) -> bool:
|
||
return isinstance(url, str) and url.startswith("data:application/pdf;base64,")
|
||
|
||
|
||
def build_user_content(
|
||
text: Optional[str], attachments: Optional[list[dict]] = None
|
||
) -> Any:
|
||
"""Return `str` (no attachments) or a list of OpenAI content-parts (with attachments).
|
||
|
||
Each attachment is `{"kind": "image"|"pdf"|"text", "name"?, "data_url"? (image/pdf),
|
||
"text"? (text)}`.
|
||
Invalid/oversized attachments are skipped rather than failing the turn.
|
||
"""
|
||
text = (text or "").strip()
|
||
attachments = attachments or []
|
||
if not attachments:
|
||
return text
|
||
|
||
parts: list[dict[str, Any]] = []
|
||
if text:
|
||
parts.append({"type": "text", "text": text})
|
||
|
||
added = 0 # attachment parts that actually made it in
|
||
for a in attachments[:MAX_ATTACHMENTS]:
|
||
if not isinstance(a, dict):
|
||
continue
|
||
kind = a.get("kind")
|
||
if kind == "image":
|
||
url = a.get("data_url") or ""
|
||
if _is_data_image(url) and len(url) <= MAX_IMAGE_CHARS:
|
||
parts.append({"type": "image_url", "image_url": {"url": url}})
|
||
added += 1
|
||
elif kind == "pdf":
|
||
url = a.get("data_url") or ""
|
||
if _is_data_pdf(url) and len(url) <= MAX_PDF_CHARS:
|
||
name = str(a.get("name") or "attachment.pdf")
|
||
parts.append(
|
||
{"type": "file", "file": {"filename": name, "file_data": url}}
|
||
)
|
||
added += 1
|
||
elif kind == "text":
|
||
body = str(a.get("text") or "")[:MAX_TEXT_CHARS]
|
||
name = str(a.get("name") or "attachment")
|
||
if body:
|
||
parts.append(
|
||
{"type": "text", "text": f"[Attached file: {name}]\n{body}"}
|
||
)
|
||
added += 1
|
||
|
||
if added == 0:
|
||
return text # every attachment was invalid/empty → just the text (possibly "")
|
||
return parts
|
||
|
||
|
||
def content_to_text(content: Any, *, image_placeholder: str = "[image]") -> str:
|
||
"""Flatten message content (string or parts) to text — for titles, previews, search.
|
||
Images render as `image_placeholder` (pass "" to drop them, e.g. for clean titles).
|
||
"""
|
||
if isinstance(content, str):
|
||
return content
|
||
if isinstance(content, list):
|
||
out = []
|
||
for part in content:
|
||
if not isinstance(part, dict):
|
||
continue
|
||
if part.get("type") == "text":
|
||
out.append(str(part.get("text", "")))
|
||
elif part.get("type") == "image_url" and image_placeholder:
|
||
out.append(image_placeholder)
|
||
elif part.get("type") == "file" and image_placeholder:
|
||
out.append("[pdf]")
|
||
return " ".join(out).strip()
|
||
return ""
|