Optimize model selection for cost-efficiency on structured extraction
The task profile is search tool invocation + JSON extraction — not reasoning or creative work. Mini models handle this equally well at 3-5x lower cost per call. OpenAI changes: - Rename is_mainline_openai_model -> is_search_capable_model - Include mini variants (gpt-5-mini, gpt-4.1-mini) in candidate pool - Exclude gpt-4o-mini (no domain filtering) and nano (no web_search) - select_openai_model() now prefers mini within newest generation - OPENAI_FALLBACK_MODELS: gpt-5-mini first, mainline as last resort - MODEL_FALLBACK_ORDER: same mini-first ordering xAI changes: - Switch alias from grok-4-1-fast (reasoning) to grok-4-1-fast-non-reasoning — same token price, faster response, no wasted reasoning tokens for structured extraction Cost per Reddit search call: ~$0.015 (gpt-5-mini) vs ~$0.044 (gpt-4.1)
This commit is contained in:
+95
-29
@@ -28,21 +28,47 @@ class TestParseVersion(unittest.TestCase):
|
||||
self.assertIsNone(result)
|
||||
|
||||
|
||||
class TestIsMainlineOpenAIModel(unittest.TestCase):
|
||||
def test_gpt5_is_mainline(self):
|
||||
class TestIsSearchCapableModel(unittest.TestCase):
|
||||
def test_gpt5_is_capable(self):
|
||||
self.assertTrue(models.is_search_capable_model("gpt-5"))
|
||||
|
||||
def test_gpt52_is_capable(self):
|
||||
self.assertTrue(models.is_search_capable_model("gpt-5.2"))
|
||||
|
||||
def test_gpt5_mini_is_capable(self):
|
||||
self.assertTrue(models.is_search_capable_model("gpt-5-mini"))
|
||||
|
||||
def test_gpt41_mini_is_capable(self):
|
||||
self.assertTrue(models.is_search_capable_model("gpt-4.1-mini"))
|
||||
|
||||
def test_gpt4o_is_capable(self):
|
||||
self.assertTrue(models.is_search_capable_model("gpt-4o"))
|
||||
|
||||
def test_gpt4o_mini_not_capable(self):
|
||||
"""gpt-4o-mini does not support web_search with domain filtering."""
|
||||
self.assertFalse(models.is_search_capable_model("gpt-4o-mini"))
|
||||
|
||||
def test_nano_not_capable(self):
|
||||
"""nano models don't support web_search."""
|
||||
self.assertFalse(models.is_search_capable_model("gpt-4.1-nano"))
|
||||
self.assertFalse(models.is_search_capable_model("gpt-5-nano"))
|
||||
|
||||
def test_gpt4_not_capable(self):
|
||||
self.assertFalse(models.is_search_capable_model("gpt-4"))
|
||||
|
||||
def test_codex_not_capable(self):
|
||||
self.assertFalse(models.is_search_capable_model("gpt-5.1-codex"))
|
||||
|
||||
def test_backward_compat_alias(self):
|
||||
"""is_mainline_openai_model still works as alias."""
|
||||
self.assertTrue(models.is_mainline_openai_model("gpt-5"))
|
||||
|
||||
def test_gpt52_is_mainline(self):
|
||||
self.assertTrue(models.is_mainline_openai_model("gpt-5.2"))
|
||||
|
||||
def test_gpt5_mini_is_not_mainline(self):
|
||||
self.assertFalse(models.is_mainline_openai_model("gpt-5-mini"))
|
||||
|
||||
def test_gpt4_is_not_mainline(self):
|
||||
self.assertFalse(models.is_mainline_openai_model("gpt-4"))
|
||||
|
||||
|
||||
class TestSelectOpenAIModel(unittest.TestCase):
|
||||
def setUp(self):
|
||||
from lib import cache
|
||||
cache.MODEL_CACHE_FILE.unlink(missing_ok=True)
|
||||
|
||||
def test_pinned_policy(self):
|
||||
result = models.select_openai_model(
|
||||
"fake-key",
|
||||
@@ -51,20 +77,8 @@ class TestSelectOpenAIModel(unittest.TestCase):
|
||||
)
|
||||
self.assertEqual(result, "gpt-5.1")
|
||||
|
||||
def test_auto_with_mock_models(self):
|
||||
mock_models = [
|
||||
{"id": "gpt-5.2", "created": 1704067200},
|
||||
{"id": "gpt-5.1", "created": 1701388800},
|
||||
{"id": "gpt-5", "created": 1698710400},
|
||||
]
|
||||
result = models.select_openai_model(
|
||||
"fake-key",
|
||||
policy="auto",
|
||||
mock_models=mock_models
|
||||
)
|
||||
self.assertEqual(result, "gpt-5.2")
|
||||
|
||||
def test_auto_filters_variants(self):
|
||||
def test_prefers_mini_over_mainline(self):
|
||||
"""Mini models should be preferred for cost-efficiency."""
|
||||
mock_models = [
|
||||
{"id": "gpt-5.2", "created": 1704067200},
|
||||
{"id": "gpt-5-mini", "created": 1704067200},
|
||||
@@ -75,8 +89,50 @@ class TestSelectOpenAIModel(unittest.TestCase):
|
||||
policy="auto",
|
||||
mock_models=mock_models
|
||||
)
|
||||
self.assertEqual(result, "gpt-5-mini")
|
||||
|
||||
def test_prefers_newer_generation_mini(self):
|
||||
"""gpt-5-mini should beat gpt-4.1-mini (newer generation)."""
|
||||
mock_models = [
|
||||
{"id": "gpt-4.1-mini", "created": 1701388800},
|
||||
{"id": "gpt-5-mini", "created": 1704067200},
|
||||
{"id": "gpt-4.1", "created": 1698710400},
|
||||
]
|
||||
result = models.select_openai_model(
|
||||
"fake-key",
|
||||
policy="auto",
|
||||
mock_models=mock_models
|
||||
)
|
||||
self.assertEqual(result, "gpt-5-mini")
|
||||
|
||||
def test_falls_back_to_mainline_when_no_mini(self):
|
||||
"""Without mini models, mainline models are selected."""
|
||||
mock_models = [
|
||||
{"id": "gpt-5.2", "created": 1704067200},
|
||||
{"id": "gpt-4.1", "created": 1698710400},
|
||||
]
|
||||
result = models.select_openai_model(
|
||||
"fake-key",
|
||||
policy="auto",
|
||||
mock_models=mock_models
|
||||
)
|
||||
self.assertEqual(result, "gpt-5.2")
|
||||
|
||||
def test_filters_unsupported_variants(self):
|
||||
"""Nano, codex, preview models should be excluded."""
|
||||
mock_models = [
|
||||
{"id": "gpt-5-nano", "created": 1704067200},
|
||||
{"id": "gpt-5.1-codex", "created": 1704067200},
|
||||
{"id": "gpt-4o-mini", "created": 1704067200},
|
||||
{"id": "gpt-4.1-mini", "created": 1698710400},
|
||||
]
|
||||
result = models.select_openai_model(
|
||||
"fake-key",
|
||||
policy="auto",
|
||||
mock_models=mock_models
|
||||
)
|
||||
self.assertEqual(result, "gpt-4.1-mini")
|
||||
|
||||
|
||||
class TestSelectXAIModel(unittest.TestCase):
|
||||
def test_latest_policy(self):
|
||||
@@ -106,6 +162,10 @@ class TestSelectXAIModel(unittest.TestCase):
|
||||
|
||||
|
||||
class TestGetModels(unittest.TestCase):
|
||||
def setUp(self):
|
||||
from lib import cache
|
||||
cache.MODEL_CACHE_FILE.unlink(missing_ok=True)
|
||||
|
||||
def test_no_keys_returns_none(self):
|
||||
config = {}
|
||||
result = models.get_models(config)
|
||||
@@ -114,9 +174,12 @@ class TestGetModels(unittest.TestCase):
|
||||
|
||||
def test_openai_key_only(self):
|
||||
config = {"OPENAI_API_KEY": "sk-test"}
|
||||
mock_models = [{"id": "gpt-5.2", "created": 1704067200}]
|
||||
mock_models = [
|
||||
{"id": "gpt-5.2", "created": 1704067200},
|
||||
{"id": "gpt-5-mini", "created": 1704067200},
|
||||
]
|
||||
result = models.get_models(config, mock_openai_models=mock_models)
|
||||
self.assertEqual(result["openai"], "gpt-5.2")
|
||||
self.assertEqual(result["openai"], "gpt-5-mini")
|
||||
self.assertIsNone(result["xai"])
|
||||
|
||||
def test_both_keys(self):
|
||||
@@ -124,10 +187,13 @@ class TestGetModels(unittest.TestCase):
|
||||
"OPENAI_API_KEY": "sk-test",
|
||||
"XAI_API_KEY": "xai-test",
|
||||
}
|
||||
mock_openai = [{"id": "gpt-5.2", "created": 1704067200}]
|
||||
mock_openai = [
|
||||
{"id": "gpt-5.2", "created": 1704067200},
|
||||
{"id": "gpt-5-mini", "created": 1704067200},
|
||||
]
|
||||
mock_xai = [{"id": "grok-4-1-fast-non-reasoning", "created": 1704067200}]
|
||||
result = models.get_models(config, mock_openai, mock_xai)
|
||||
self.assertEqual(result["openai"], "gpt-5.2")
|
||||
self.assertEqual(result["openai"], "gpt-5-mini")
|
||||
self.assertEqual(result["xai"], "grok-4-1-fast-non-reasoning")
|
||||
|
||||
|
||||
|
||||
@@ -64,13 +64,18 @@ class TestIsModelAccessError(unittest.TestCase):
|
||||
class TestModelFallbackOrder(unittest.TestCase):
|
||||
"""Tests for MODEL_FALLBACK_ORDER constant."""
|
||||
|
||||
def test_contains_gpt4o(self):
|
||||
"""Fallback list should include gpt-4o."""
|
||||
def test_mini_first(self):
|
||||
"""Mini models should come first (cost-efficient for structured extraction)."""
|
||||
self.assertEqual(MODEL_FALLBACK_ORDER[0], "gpt-5-mini")
|
||||
|
||||
def test_contains_mainline_fallbacks(self):
|
||||
"""Fallback list should include mainline models as last resort."""
|
||||
self.assertIn("gpt-4.1", MODEL_FALLBACK_ORDER)
|
||||
self.assertIn("gpt-4o", MODEL_FALLBACK_ORDER)
|
||||
|
||||
def test_gpt41_is_first(self):
|
||||
"""gpt-4.1 should be the first fallback option."""
|
||||
self.assertEqual(MODEL_FALLBACK_ORDER[0], "gpt-4.1")
|
||||
def test_no_gpt4o_mini(self):
|
||||
"""gpt-4o-mini should NOT be in fallback (no domain filtering support)."""
|
||||
self.assertNotIn("gpt-4o-mini", MODEL_FALLBACK_ORDER)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
Reference in New Issue
Block a user