diff --git a/libs/partners/openai/langchain_openai/chat_models/azure.py b/libs/partners/openai/langchain_openai/chat_models/azure.py index f429ff25a4..cdb022d2ae 100644 --- a/libs/partners/openai/langchain_openai/chat_models/azure.py +++ b/libs/partners/openai/langchain_openai/chat_models/azure.py @@ -760,11 +760,19 @@ class AzureChatOpenAI(BaseChatOpenAI): def _resolve_model_profile(self) -> ModelProfile | None: if (self.model_name is not None) and ( - profile := _get_default_model_profile(self.model_name) or None + profile := _get_default_model_profile( + self.model_name, use_responses_api=self._use_responses_api({}) + ) + or None ): return profile if self.deployment_name is not None: - return _get_default_model_profile(self.deployment_name) or None + return ( + _get_default_model_profile( + self.deployment_name, use_responses_api=self._use_responses_api({}) + ) + or None + ) return None @property diff --git a/libs/partners/openai/langchain_openai/chat_models/base.py b/libs/partners/openai/langchain_openai/chat_models/base.py index cca86677ee..7a0a3db41a 100644 --- a/libs/partners/openai/langchain_openai/chat_models/base.py +++ b/libs/partners/openai/langchain_openai/chat_models/base.py @@ -165,6 +165,7 @@ from langchain_openai.chat_models._compat import ( _convert_to_v03_ai_message, _unwrap_non_standard, ) +from langchain_openai.data._file_mime_types import _FILE_MIME_TYPES from langchain_openai.data._profiles import _PROFILES if TYPE_CHECKING: @@ -195,9 +196,39 @@ def _get_ssrf_safe_client() -> httpx.Client: _MODEL_PROFILES = cast(ModelProfileRegistry, _PROFILES) -def _get_default_model_profile(model_name: str) -> ModelProfile: - default = _MODEL_PROFILES.get(model_name) or {} - return default.copy() +def _get_default_model_profile( + model_name: str, *, use_responses_api: bool = False +) -> ModelProfile: + profile = (_MODEL_PROFILES.get(model_name) or {}).copy() + if not profile: + return profile + if "file_mime_types" in profile: + profile["file_mime_types"] = list(profile["file_mime_types"]) + return profile + supported_modalities = { + "application/pdf": profile.get("pdf_inputs") and profile.get("image_inputs"), + "image/jpeg": profile.get("image_inputs"), + "image/png": profile.get("image_inputs"), + "image/webp": profile.get("image_inputs"), + "image/gif": profile.get("image_inputs"), + "audio/wav": profile.get("audio_inputs") and not use_responses_api, + "audio/mpeg": profile.get("audio_inputs") and not use_responses_api, + } + mime_types = [ + mime_type + for mime_type in _FILE_MIME_TYPES + if supported_modalities.get( + mime_type, + use_responses_api + and profile.get("tool_calling") + and not profile.get("audio_outputs"), + ) + ] + if mime_types and profile.get("text_outputs") and not profile.get("image_outputs"): + profile["file_mime_types"] = mime_types + else: + profile.pop("file_mime_types", None) + return profile WellKnownTools = ( @@ -1545,7 +1576,12 @@ class BaseChatOpenAI(BaseChatModel): return self def _resolve_model_profile(self) -> ModelProfile | None: - return _get_default_model_profile(self.model_name) or None + return ( + _get_default_model_profile( + self.model_name, use_responses_api=self._use_responses_api({}) + ) + or None + ) @property def _default_params(self) -> dict[str, Any]: diff --git a/libs/partners/openai/langchain_openai/data/_file_mime_types.py b/libs/partners/openai/langchain_openai/data/_file_mime_types.py new file mode 100644 index 0000000000..cd4a3c0ba8 --- /dev/null +++ b/libs/partners/openai/langchain_openai/data/_file_mime_types.py @@ -0,0 +1,137 @@ +"""Provider file MIME type defaults and backend restrictions.""" + +_FILE_MIME_TYPES = ( + "application/pdf", + "image/jpeg", + "image/png", + "image/webp", + "image/gif", + "audio/wav", + "audio/mpeg", + "application/csv", + "application/graphql", + "application/javascript", + "application/json", + "application/json5", + "application/msword", + "application/rtf", + "application/toml", + "application/typescript", + "application/vnd.apple.iwork", + "application/vnd.apple.keynote", + "application/vnd.apple.pages", + "application/vnd.google-apps.document", + "application/vnd.google-apps.presentation", + "application/vnd.google-apps.spreadsheet", + "application/vnd.ms-excel", + "application/vnd.ms-powerpoint", + "application/vnd.oasis.opendocument.text", + "application/vnd.openxmlformats-officedocument.presentationml.presentation", + "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", + "application/vnd.openxmlformats-officedocument.wordprocessingml.document", + "application/x-awk", + "application/x-bash", + "application/x-graphql", + "application/x-httpd-php", + "application/x-httpd-php-source", + "application/x-iif", + "application/x-json5", + "application/x-ndjson", + "application/x-patch", + "application/x-php", + "application/x-powershell", + "application/x-protobuf", + "application/x-rust", + "application/x-scala", + "application/x-sql", + "application/x-subrip", + "application/x-terraform", + "application/x-toml", + "application/x-yaml", + "application/yaml", + "message/rfc822", + "text/calendar", + "text/css", + "text/csv", + "text/html", + "text/javascript", + "text/jsx", + "text/markdown", + "text/plain", + "text/rtf", + "text/srt", + "text/tsv", + "text/tsx", + "text/vbscript", + "text/vtt", + "text/x-R", + "text/x-asm", + "text/x-astro", + "text/x-awk", + "text/x-bash", + "text/x-c", + "text/x-c++", + "text/x-clojure", + "text/x-cmake", + "text/x-csharp", + "text/x-dart", + "text/x-diff", + "text/x-dockerfile", + "text/x-ejs", + "text/x-elixir", + "text/x-erb", + "text/x-erlang", + "text/x-go", + "text/x-golang", + "text/x-gradle", + "text/x-graphql", + "text/x-groovy", + "text/x-handlebars", + "text/x-haskell", + "text/x-hcl", + "text/x-iif", + "text/x-ini", + "text/x-jade", + "text/x-java", + "text/x-jinja2", + "text/x-julia", + "text/x-kotlin", + "text/x-less", + "text/x-liquid", + "text/x-lisp", + "text/x-lua", + "text/x-makefile", + "text/x-mustache", + "text/x-objectivec", + "text/x-objectivec++", + "text/x-patch", + "text/x-perl", + "text/x-php", + "text/x-properties", + "text/x-protobuf", + "text/x-pug", + "text/x-python", + "text/x-r", + "text/x-rst", + "text/x-ruby", + "text/x-rust", + "text/x-sass", + "text/x-scala", + "text/x-script.python", + "text/x-scss", + "text/x-sh", + "text/x-shellscript", + "text/x-sql", + "text/x-subrip", + "text/x-swift", + "text/x-terraform", + "text/x-tex", + "text/x-tmpl", + "text/x-toml", + "text/x-twig", + "text/x-typescript", + "text/x-vcard", + "text/x-yaml", + "text/x-zsh", + "text/xml", +) diff --git a/libs/partners/openai/langchain_openai/data/_profiles.py b/libs/partners/openai/langchain_openai/data/_profiles.py index c14effd36d..e02f8a2cf2 100644 --- a/libs/partners/openai/langchain_openai/data/_profiles.py +++ b/libs/partners/openai/langchain_openai/data/_profiles.py @@ -41,6 +41,7 @@ _PROFILES: dict[str, dict[str, Any]] = { "image_tool_message": True, "tool_choice": True, "tool_call_streaming": True, + "file_mime_types": [], }, "gpt-3.5-turbo": { "name": "GPT-3.5-turbo", @@ -69,6 +70,7 @@ _PROFILES: dict[str, dict[str, Any]] = { "image_tool_message": False, "tool_choice": True, "tool_call_streaming": True, + "file_mime_types": [], }, "gpt-4": { "name": "GPT-4", @@ -97,6 +99,7 @@ _PROFILES: dict[str, dict[str, Any]] = { "image_tool_message": True, "tool_choice": True, "tool_call_streaming": True, + "file_mime_types": [], }, "gpt-4-turbo": { "name": "GPT-4 Turbo", @@ -1383,6 +1386,7 @@ _PROFILES: dict[str, dict[str, Any]] = { "image_tool_message": True, "tool_choice": True, "tool_call_streaming": True, + "file_mime_types": [], }, "gpt-image-1-mini": { "name": "gpt-image-1-mini", @@ -1409,6 +1413,7 @@ _PROFILES: dict[str, dict[str, Any]] = { "image_tool_message": True, "tool_choice": True, "tool_call_streaming": True, + "file_mime_types": [], }, "gpt-image-1.5": { "name": "gpt-image-1.5", @@ -1435,6 +1440,7 @@ _PROFILES: dict[str, dict[str, Any]] = { "image_tool_message": True, "tool_choice": True, "tool_call_streaming": True, + "file_mime_types": [], }, "gpt-image-2": { "name": "gpt-image-2", @@ -1461,6 +1467,7 @@ _PROFILES: dict[str, dict[str, Any]] = { "image_tool_message": True, "tool_choice": True, "tool_call_streaming": True, + "file_mime_types": [], }, "gpt-realtime-2.1": { "name": "GPT-Realtime-2.1", @@ -1488,6 +1495,7 @@ _PROFILES: dict[str, dict[str, Any]] = { "image_tool_message": True, "tool_choice": True, "tool_call_streaming": True, + "file_mime_types": [], }, "o1": { "name": "o1", @@ -1680,6 +1688,7 @@ _PROFILES: dict[str, dict[str, Any]] = { "image_tool_message": True, "tool_choice": True, "tool_call_streaming": True, + "file_mime_types": [], }, "text-embedding-3-small": { "name": "text-embedding-3-small", @@ -1706,6 +1715,7 @@ _PROFILES: dict[str, dict[str, Any]] = { "image_tool_message": True, "tool_choice": True, "tool_call_streaming": True, + "file_mime_types": [], }, "text-embedding-ada-002": { "name": "text-embedding-ada-002", @@ -1732,5 +1742,6 @@ _PROFILES: dict[str, dict[str, Any]] = { "image_tool_message": True, "tool_choice": True, "tool_call_streaming": True, + "file_mime_types": [], }, } diff --git a/libs/partners/openai/langchain_openai/data/profile_augmentations.toml b/libs/partners/openai/langchain_openai/data/profile_augmentations.toml index 3b8a885c8e..dc44738fbe 100644 --- a/libs/partners/openai/langchain_openai/data/profile_augmentations.toml +++ b/libs/partners/openai/langchain_openai/data/profile_augmentations.toml @@ -8,7 +8,38 @@ image_tool_message = true tool_choice = true tool_call_streaming = true +[overrides."chatgpt-image-latest"] +file_mime_types = [] + +[overrides."gpt-image-1"] +file_mime_types = [] + +[overrides."gpt-image-1-mini"] +file_mime_types = [] + +[overrides."gpt-image-1.5"] +file_mime_types = [] + +[overrides."gpt-image-2"] +file_mime_types = [] + +[overrides."gpt-realtime-2.1"] +file_mime_types = [] + +[overrides."text-embedding-3-large"] +file_mime_types = [] + +[overrides."text-embedding-3-small"] +file_mime_types = [] + +[overrides."text-embedding-ada-002"] +file_mime_types = [] + +[overrides."gpt-4"] +file_mime_types = [] + [overrides."gpt-3.5-turbo"] +file_mime_types = [] image_url_inputs = false pdf_inputs = false pdf_tool_message = false diff --git a/libs/partners/openai/tests/unit_tests/chat_models/test_file_mime_types.py b/libs/partners/openai/tests/unit_tests/chat_models/test_file_mime_types.py new file mode 100644 index 0000000000..f437075a56 --- /dev/null +++ b/libs/partners/openai/tests/unit_tests/chat_models/test_file_mime_types.py @@ -0,0 +1,203 @@ +"""Test instance-level file MIME type capabilities.""" + +from copy import deepcopy +from typing import Any + +import pytest +from langchain_core.language_models import ModelProfile +from langchain_core.runnables import RunnableBinding +from pydantic import SecretStr + +from langchain_openai import AzureChatOpenAI, ChatOpenAI +from langchain_openai.chat_models.base import _get_default_model_profile +from langchain_openai.data._profiles import _PROFILES + + +@pytest.mark.parametrize( + ("model_name", "kwargs", "generic_files"), + [ + ("gpt-4.1", {}, False), + ("gpt-4.1", {"use_responses_api": True}, True), + ("gpt-4.1", {"reasoning": {"summary": "auto"}}, True), + ("gpt-4.1", {"output_version": "responses/v1"}, True), + ("gpt-5.2-pro", {}, True), + ("gpt-5.2-pro", {"use_responses_api": False}, False), + ("gpt-4.1", {"use_responses_api": False, "reasoning": {}}, False), + ], +) +def test_file_mime_types_routing( + model_name: str, kwargs: dict[str, Any], *, generic_files: bool +) -> None: + model = ChatOpenAI(model=model_name, api_key=SecretStr("test"), **kwargs) + assert model.profile + mime_types = model.profile["file_mime_types"] + assert "application/pdf" in mime_types + assert {"image/jpeg", "image/png", "image/webp", "image/gif"} <= set(mime_types) + assert ("text/plain" in mime_types) is generic_files + assert ("text/csv" in mime_types) is generic_files + assert not any( + mime_type.startswith(("audio/", "video/")) for mime_type in mime_types + ) + + +@pytest.mark.parametrize( + "model_name", + [ + "text-embedding-ada-002", + "text-embedding-3-small", + "text-embedding-3-large", + "chatgpt-image-latest", + "gpt-image-1", + "gpt-image-1-mini", + "gpt-image-1.5", + "gpt-image-2", + "gpt-realtime-2.1", + ], +) +def test_non_responses_models_omit_file_mime_types(model_name: str) -> None: + model = ChatOpenAI( + model=model_name, api_key=SecretStr("test"), use_responses_api=True + ) + assert (model.profile or {})["file_mime_types"] == [] + assert _PROFILES[model_name]["file_mime_types"] == [] + + +def test_file_mime_types_isolation() -> None: + expected = deepcopy(_PROFILES["gpt-4.1"]) + first = ChatOpenAI( + model="gpt-4.1", api_key=SecretStr("test"), use_responses_api=True + ) + assert first.profile + first.profile["file_mime_types"].clear() + second = ChatOpenAI( + model="gpt-4.1", api_key=SecretStr("test"), use_responses_api=True + ) + completions = ChatOpenAI(model="gpt-4.1", api_key=SecretStr("test")) + assert second.profile + assert "text/plain" in second.profile["file_mime_types"] + assert completions.profile + assert "application/pdf" in completions.profile["file_mime_types"] + assert "text/plain" not in completions.profile["file_mime_types"] + assert _PROFILES["gpt-4.1"] == expected + + +@pytest.mark.parametrize("use_responses_api", [False, True]) +@pytest.mark.parametrize("model_name", ["unknown-model", "gpt-image-2"]) +def test_explicit_file_mime_types( + monkeypatch: pytest.MonkeyPatch, model_name: str, *, use_responses_api: bool +) -> None: + profile: ModelProfile = {"file_mime_types": ["application/custom"]} + model = ChatOpenAI( + model=model_name, + api_key=SecretStr("test"), + use_responses_api=use_responses_api, + profile=profile, + ) + assert model.profile == profile + monkeypatch.setitem(_PROFILES, model_name, profile) + inferred = _get_default_model_profile( + model_name, use_responses_api=use_responses_api + ) + assert inferred == profile + inferred["file_mime_types"].clear() + assert profile["file_mime_types"] == ["application/custom"] + + +@pytest.mark.parametrize("use_responses_api", [False, True]) +@pytest.mark.parametrize("model_name", [None, "gpt-4.1", "unknown-model"]) +def test_azure_file_mime_types( + model_name: str | None, *, use_responses_api: bool +) -> None: + model = AzureChatOpenAI( + model=model_name, + azure_deployment="gpt-4.1", + azure_endpoint="https://example.openai.azure.com", + api_version="2025-04-01-preview", + api_key=SecretStr("test"), + use_responses_api=use_responses_api, + ) + assert model.profile + assert "application/pdf" in model.profile["file_mime_types"] + assert ("text/plain" in model.profile["file_mime_types"]) is use_responses_api + + +def test_per_call_routing_does_not_change_profile() -> None: + model = ChatOpenAI(model="gpt-4.1", api_key=SecretStr("test")) + bound = model.bind_tools([{"type": "web_search"}]) + assert isinstance(bound, RunnableBinding) + assert model._use_responses_api(dict(bound.kwargs)) + assert model.profile + assert "text/plain" not in model.profile["file_mime_types"] + + +@pytest.mark.parametrize("use_responses_api", [False, True]) +def test_unknown_and_text_only_models_omit_file_mime_types( + *, use_responses_api: bool +) -> None: + for model_name in ("unknown-model", "gpt-3.5-turbo", "gpt-4"): + profile = _get_default_model_profile( + model_name, use_responses_api=use_responses_api + ) + assert profile.get("file_mime_types", []) == [] + + +@pytest.mark.parametrize( + ("flags", "use_responses_api", "expected"), + [ + ( + {"pdf_inputs": False, "image_inputs": True}, + False, + ["image/jpeg", "image/png", "image/webp", "image/gif"], + ), + ({"pdf_inputs": True, "image_inputs": False}, False, []), + ({"pdf_inputs": True, "image_inputs": False}, True, ["text/plain"]), + ({"audio_inputs": True}, False, ["audio/wav", "audio/mpeg"]), + ({"audio_inputs": True}, True, ["text/plain"]), + ( + {"audio_inputs": True, "audio_outputs": True}, + False, + ["audio/wav", "audio/mpeg"], + ), + ({"audio_inputs": True, "audio_outputs": True}, True, []), + ({"image_inputs": True, "image_outputs": True}, False, []), + ( + {"image_inputs": True, "tool_calling": False}, + True, + ["image/jpeg", "image/png", "image/webp", "image/gif"], + ), + ({"audio_inputs": False, "video_inputs": True}, False, []), + ], +) +def test_file_mime_types_follow_modalities_and_transport( + monkeypatch: pytest.MonkeyPatch, + flags: dict[str, bool], + *, + use_responses_api: bool, + expected: list[str], +) -> None: + mime_types = [ + "application/pdf", + "image/jpeg", + "image/png", + "image/webp", + "image/gif", + "audio/wav", + "audio/mpeg", + "text/plain", + ] + profiles = { + "synthetic": { + "text_outputs": True, + "tool_calling": True, + **flags, + } + } + monkeypatch.setattr("langchain_openai.chat_models.base._MODEL_PROFILES", profiles) + monkeypatch.setattr( + "langchain_openai.chat_models.base._FILE_MIME_TYPES", mime_types + ) + profile = _get_default_model_profile( + "synthetic", use_responses_api=use_responses_api + ) + assert profile.get("file_mime_types", []) == expected + assert "file_mime_types" not in profiles["synthetic"]