From e80b9109e68e81b4889b2ca5e05d3fd52dbbecf0 Mon Sep 17 00:00:00 2001 From: "J.Jason" <130959319+JJasonSun@users.noreply.github.com> Date: Fri, 2 Oct 2026 22:45:28 +0800 Subject: [PATCH] feat: update ECNU multimodal retrieval and image editing guidance --- README.md | 2 +- SKILL.md | 4 +- references/agent_development.md | 6 ++ references/api_reference.md | 7 +- references/examples.md | 133 +++++++++++++++++++++++++++-- references/known_deviations.md | 5 ++ references/models.md | 28 ++++++ references/workflows.md | 11 ++- scripts/smoke_test.py | 45 +++++----- tests/test_repository_contracts.py | 99 +++++++++++++++++++++ tests/test_smoke_test.py | 124 +++++++++++++++++++-------- 11 files changed, 391 insertions(+), 73 deletions(-) diff --git a/README.md b/README.md index ed3f10d..58563cd 100644 --- a/README.md +++ b/README.md @@ -37,7 +37,7 @@ credentials should not prevent the Agent from writing or reviewing code. |---|---| | Start a normal integration | [SKILL.md](SKILL.md) | | Find the relevant official endpoint contract | [Endpoint map](references/api_reference.md) | -| LangChain embeddings, thinking/tool history, Anthropic SDK | [Integration recipes](references/examples.md) | +| Image edits, multimodal retrieval, LangChain embeddings, thinking/tool history, Anthropic SDK | [Integration recipes](references/examples.md) | | Model choice or current account facts | [Model/account pointers](references/models.md) | | Diagnose a failure or select a live probe | [Targeted diagnosis](references/workflows.md) | | Inspect dated evidence | [Known deviations](references/known_deviations.md) | diff --git a/SKILL.md b/SKILL.md index e409eb7..253efbf 100644 --- a/SKILL.md +++ b/SKILL.md @@ -3,7 +3,7 @@ name: ecnu-api description: > Connect an application or SDK to the ECNU / ChatECNU LLM Open Platform at chat.ecnu.edu.cn, or diagnose an ECNU API failure. Covers chat, - Responses, tools, vision, embeddings, rerank, images, TTS, and + Responses, tools, vision, text/image retrieval, image generation/editing, TTS, and Anthropic-compatible clients. Not for general ECNU information, unrelated model questions, or generic Agent/prompt design. --- @@ -55,6 +55,8 @@ Its full messages URL ends in `/open/api/anthropic/v1/messages`. |---|---| | Chat, streaming, vision, Responses, JSON, rerank, images, TTS, or browser integration | [Official endpoint map](references/api_reference.md): open only the matching page | | LangChain embeddings | [Embedding recipe](references/examples.md#langchain-embeddings) | +| Image editing or generation | [Image recipe](references/examples.md#image-generation-and-editing): distinguish JSON generation from multipart editing | +| Image/text embeddings or rerank | [Multimodal retrieval](references/examples.md#multimodal-retrieval): choose the model and preserve vector-space compatibility | | Anthropic SDK setup | [Anthropic recipe](references/examples.md#anthropic-sdk) | | Thinking and tool continuation | [Tool-history recipe](references/examples.md#thinking-and-tool-history) plus the linked ECNU contract | | Model choice, prices, quotas, or deployment questions | [Model and account pointers](references/models.md) | diff --git a/references/agent_development.md b/references/agent_development.md index 36a302d..ee96309 100644 --- a/references/agent_development.md +++ b/references/agent_development.md @@ -8,6 +8,12 @@ Model facts and protocol rules are linked from [models.md](models.md) and [api_reference.md](api_reference.md). The design advice and prompt examples below are **`application-policy`**: starting points to evaluate on your tasks. +**Contract update checked 2026-10-02:** the 2026-09-18 model change made +`ecnu-max` text-only again; use `ecnu-plus` for vision. Plus now supports +`low` / `medium` / `xhigh` effort when thinking is enabled. The older model +choices below describe the September 12 checks, not current execution advice; +follow [current model selection](models.md) and [thinking](examples.md#thinking-and-tool-history). + ## Choose a model and reasoning budget | Workload | Starting point | What to check before escalating | diff --git a/references/api_reference.md b/references/api_reference.md index acf2bb4..3ff7ed3 100644 --- a/references/api_reference.md +++ b/references/api_reference.md @@ -23,9 +23,10 @@ in environment variables; ticket URLs are also credentials. | Chat, streaming, tools, image understanding | [Chat Completions](https://developer.ecnu.edu.cn/vitepress/llm/api/completions.html) | [Thinking/tool history](examples.md#thinking-and-tool-history) | | Responses-compatible client | [Responses](https://developer.ecnu.edu.cn/vitepress/llm/api/responses.html) | Verify the specific advanced tool/event support; compatibility alone is insufficient | | JSON Schema or JSON object output | [Structured output](https://developer.ecnu.edu.cn/vitepress/llm/api/structuredoutput.html) | Check completion, parse raw JSON, and validate the supplied schema; do not hide a mismatch by stripping fences | -| Embeddings | [Text vectors](https://developer.ecnu.edu.cn/vitepress/llm/api/embedding.html) | [Raw-string LangChain recipe](examples.md#langchain-embeddings) | -| Rerank | [Rerank](https://developer.ecnu.edu.cn/vitepress/llm/api/rerank.html) | Use the endpoint's contract, not an assumed OpenAI SDK method | -| Image generation | [Images](https://developer.ecnu.edu.cn/vitepress/llm/api/imagegenerate.html) | Observe the current URL lifetime and avoid duplicate paid generations | +| Text or multimodal embeddings | [Vectors](https://developer.ecnu.edu.cn/vitepress/llm/api/embedding.html) | [Text LangChain recipe](examples.md#langchain-embeddings) or [VL payloads](examples.md#multimodal-retrieval); dimensions are model-specific | +| Text or multimodal rerank | [Rerank](https://developer.ecnu.edu.cn/vitepress/llm/api/rerank.html) | [VL payloads](examples.md#multimodal-retrieval); use the endpoint's contract, not an assumed OpenAI SDK method | +| Image generation | [Images](https://developer.ecnu.edu.cn/vitepress/llm/api/imagegenerate.html) | [Generation/edit differences](examples.md#image-generation-and-editing); original and revised prompts can differ | +| Edit an existing image | [Image edits](https://developer.ecnu.edu.cn/vitepress/llm/api/imageedit.html) | [Multipart recipe](examples.md#image-generation-and-editing); one uploaded image, original instruction, no output-size control | | Text-to-speech | [Audio](https://developer.ecnu.edu.cn/vitepress/llm/api/audio.html) | [Non-JSON errors and PCM headers](workflows.md#start-from-the-symptom) | | Model discovery | [Models endpoint](https://developer.ecnu.edu.cn/vitepress/llm/api/models.html) | A visible ID does not prove usable capability or valid authentication | | Anthropic-compatible client | [Anthropic API](https://developer.ecnu.edu.cn/vitepress/llm/api/anthropic.html) | [SDK setup](examples.md#anthropic-sdk); investigate suffix errors only when they occur | diff --git a/references/examples.md b/references/examples.md index bce77f2..c26cb08 100644 --- a/references/examples.md +++ b/references/examples.md @@ -14,13 +14,13 @@ an example edit alone does not refresh that evidence. ## LangChain embeddings -The [ECNU embedding contract](https://developer.ecnu.edu.cn/vitepress/llm/api/embedding.html) +For `ecnu-embedding-small`, the [ECNU embedding contract](https://developer.ecnu.edu.cn/vitepress/llm/api/embedding.html) accepts strings, not OpenAI token-ID arrays. Disable LangChain's token conversion. -The official LangChain example sets `dimensions` to 1024, but the request table -lists only `model` and `input` and does not explain dimension selection. -This recipe omits `dimensions` and validates the documented 1024-value output, -following the dated recipe coverage above. That check does not establish whether -the service accepts or rejects the field. Reject an empty list before making a call. +The current parameter table limits `dimensions` to `ecnu-embedding-vl`, although +the page's older LangChain example still sets it for the small model. This recipe +follows the parameter table: omit `dimensions` and validate the small model's +1024-value output. Reject an empty list before making a call. Multimodal inputs +use the [VL recipe](#multimodal-retrieval), not this text-only adapter. Standalone example; requires `langchain-openai`: @@ -49,13 +49,128 @@ For longer inputs, follow the current endpoint's character limit and split locally. Do not infer an undocumented batch maximum from per-call billing. When using direct HTTP, align vectors with inputs by their returned `index`. +## Multimodal retrieval + +Use the current [embedding](https://developer.ecnu.edu.cn/vitepress/llm/api/embedding.html) +and [rerank](https://developer.ecnu.edu.cn/vitepress/llm/api/rerank.html) contracts. +Both VL models accept strings or objects with `text`, `image`, or both. An +`image` is a complete PNG/JPEG base64 data URL, not a remote URL or Chat +Completions `image_url` content block. Validate image type, decoded size +(at most 5 MiB) and pixel count (at most 16 million) before encoding/uploading. +Each object holds one image; video and multi-image objects are unsupported. + +Request-building fragment for an existing HTTP client; `image_data_url` has +already passed those checks. These dictionaries do not send requests: + +```python +item = {"text": "A red flower beside a green leaf", "image": image_data_url} +embedding_payload = { + "model": "ecnu-embedding-vl", + "input": [item, "A red flower"], + "dimensions": 1024, + "encoding_format": "float", +} +rerank_payload = { + "model": "ecnu-rerank-vl", + "query": "A red flower", + "documents": [item, "A blue car"], + "top_n": 2, + "return_documents": False, +} +``` + +Send `embedding_payload` as JSON to `POST /embeddings`, or `rerank_payload` +as JSON to `POST /rerank`, using the OpenAI-compatible base and Bearer auth. +Keep requests serial when checking both. For embedding, cap batches at 32 +items; supported dimensions are 1024/2048/4096, with 4096 as the default. +Check the returned count, unique `index` values and finite vector lengths +against the inputs and requested dimension. `encoding_format=base64` instead +returns little-endian float32 bytes encoded as a string; `usage` may be null. + +For rerank, map `results[].index` back to the original candidates and validate +finite scores in [0, 1]. `return_documents=True` can echo entire image data +URLs, so omit that output from logs. The embedding batch limit is not a +documented rerank limit. Both endpoint pages retain an 8192-character text +limit; the model page's 32K context is not permission to exceed it. + +Choose the model explicitly. Keep an existing text index on +`ecnu-embedding-small` until a deliberate migration: even a 1024-dimensional +VL vector belongs to a different space. A VL reranker only changes ordering +of candidates; it does not require a new vector index or replace image-text +consistency/education review. + +## Image generation and editing + +The [model page](https://developer.ecnu.edu.cn/vitepress/llm/model.html) lists +`ecnu-image` as qwen-image-2.1 following the 2026-09-30 upgrade. Use the stable +alias and choose the endpoint according to the task: + +- [Generation](https://developer.ecnu.edu.cn/vitepress/llm/api/imagegenerate.html) + uses JSON at `POST /images/generations`. Prompts are automatically expanded; + retain the original prompt separately from a returned `revised_prompt`. + Check this endpoint's allowed `size` values; do not assume arbitrary sizes or + a batch `n` parameter from OpenAI compatibility. +- [Editing](https://developer.ecnu.edu.cn/vitepress/llm/api/imageedit.html) + uses multipart at `POST /images/edits`, with one `image` file and a `prompt` + of at most 1024 characters. Instructions pass through unchanged after safety + review. Only `n=1` is supported; `size` is ignored and not forwarded. + +Both support `url` and `b64_json` results. URLs expire after 24 hours; transfer +them promptly when using URL output. Both add an AI watermark. A successful +HTTP status or non-empty `data` alone is insufficient: a rejected request can +have `err_message` and only a masked `revised_prompt`, without an image. +Do not infer mask, multiple references, transparency, fixed edit dimensions, +or guaranteed character consistency from the editing capability. The editing +page does not specify upload size/format limits; do not copy the VL limits. + +Direct HTTP editing fragment; `image_path` names a local image approved for +upload and `prompt` is the editing instruction. Requires `requests`; no SDK +change is needed. This makes one request with no automatic retry: + +```python +import base64 +import os +import requests + +if not isinstance(prompt, str) or not prompt.strip() or len(prompt) > 1024: + raise ValueError("Expected a non-empty editing instruction of at most 1024 characters") +with open(image_path, "rb") as source: + response = requests.post( + "https://chat.ecnu.edu.cn/open/api/v1/images/edits", + headers={"Authorization": f"Bearer {os.environ['ECNU_API_KEY']}"}, + data={"model": "ecnu-image", "prompt": prompt, "response_format": "b64_json"}, + files={"image": source}, + timeout=120, + ) +response.raise_for_status() +body = response.json() +if not isinstance(body, dict) or body.get("err_message"): + raise RuntimeError("Image edit failed; no output accepted") +items = body.get("data") +if not isinstance(items, list) or len(items) != 1 or not isinstance(items[0], dict): + raise RuntimeError("Expected one edited image") +encoded = items[0].get("b64_json") +if not isinstance(encoded, str) or not encoded: + raise RuntimeError("Image edit returned no image bytes") +edited_bytes = base64.b64decode(encoded, validate=True) +if not edited_bytes: + raise RuntimeError("Image edit returned empty image bytes") +``` + +Decode with the application's image loader before accepting or saving these +bytes, then save to a new asset path and inspect the result before replacing +the source. The fragment leaves the input file untouched. Actual edit quality, +character retention and output dimensions still require live verification. + ## Thinking and tool history Use the [ECNU thinking contract](https://developer.ecnu.edu.cn/vitepress/llm/thinking.html) and the wire format of the chosen endpoint. For Chat Completions, enable -thinking through `thinking.type`; direct `reasoning_effort` uses `low`, `high`, -or `max` on `ecnu-max` and is ignored by `ecnu-plus`. Pass ECNU extensions through -`extra_body` with the OpenAI SDK. Do not substitute upstream template switches. +thinking through `thinking.type`; direct `reasoning_effort` uses `low` / `high` / +`max` on `ecnu-max`, and `low` / `medium` / `xhigh` on `ecnu-plus`. Effort applies +when thinking is enabled; thinking defaults to off. Pass ECNU extensions through +`extra_body` with the OpenAI SDK. Do not substitute upstream template switches +or silently map an unsupported tier between models. For a tool exchange, append the complete actual assistant message before the matching tool results. ECNU documents preserving `reasoning_content` for diff --git a/references/known_deviations.md b/references/known_deviations.md index 5713273..9306a61 100644 --- a/references/known_deviations.md +++ b/references/known_deviations.md @@ -193,6 +193,11 @@ Only response structure, field-preservation checks, and usage were retained. ## Direct `ecnu-max` image input +Current-contract note (documentation checked 2026-10-02, not a retest): the +[2026-09-18 model change](https://developer.ecnu.edu.cn/vitepress/llm/model.html) +made `ecnu-max` text-only again. Use `ecnu-plus` for image understanding. +The resolved status below describes only the September 12 fixture. + - **Tested at:** 2026-09-12; previous failure on 2026-08-23 - **Environment:** live-2026-09-12-a, direct HTTP; historical live-2026-08-23-a - **Protocol and endpoint:** OpenAI-compatible `POST /chat/completions` diff --git a/references/models.md b/references/models.md index 04b0579..a24c5dd 100644 --- a/references/models.md +++ b/references/models.md @@ -17,6 +17,34 @@ a measured quality or latency ranking. | What changed and when | [Release notes](https://developer.ecnu.edu.cn/vitepress/llm/release.html) | | Is there a reported incident? | [Service status](https://chat.ecnu.edu.cn/status) | +## Select the capability, not just a new model name + +The [2026-09-30 release](https://developer.ecnu.edu.cn/vitepress/llm/release.html) +added multimodal retrieval and image editing. Contract checked 2026-10-02; +these are documented capabilities, not new live-test results. + +| Model | Backend in the current model page | Integration choice | +|---|---|---| +| `ecnu-embedding-small` | bge-m3 | Keep for existing text-only, 1024-dimensional indexes | +| `ecnu-embedding-vl` | Qwen3-VL-Embedding-8B | Text, one image per item, or both; default 4096 dimensions, optional 1024/2048/4096 | +| `ecnu-rerank` | bge-reranker-v2-m3 | Text query and text candidates | +| `ecnu-rerank-vl` | Qwen3-VL-Reranker-8B | Text/image query and candidates; rerank an already retrieved candidate set | +| `ecnu-image` | qwen-image-2.1 | Generate from text or edit one existing image; different request formats and prompt handling | + +Use the [retrieval and image recipes](examples.md) for the differences that +affect requests. Equal vector dimensions do not imply compatible embedding +spaces: changing embedding models requires re-embedding the indexed items and +using that same model and dimensions for queries. A reranker can change without +replacing stored vectors; its relevance scores are not scientific or educational +quality judgments and should not be compared across models. + +For image understanding, the current model page specifies `ecnu-plus`. +`ecnu-max` currently uses DeepSeek-V4-Flash-0731 and is text-only; older +DeepSeek-V4.1 and successful vision observations are historical, not current +capability guarantees. Keep stable ECNU aliases in requests and recheck backend +labels before displaying them. Do not relabel old generated assets as output +from a newly announced backend. + ## ECNU-specific boundaries Use primary ECNU model names for new integrations rather than upstream names diff --git a/references/workflows.md b/references/workflows.md index d34e35f..5dd043a 100644 --- a/references/workflows.md +++ b/references/workflows.md @@ -22,7 +22,7 @@ These links are dated observations, not claims that the issue still reproduces. | Tool continuation loses state or reasoning fields | Preserve the actual message and inspect serialization: [recipe](examples.md#thinking-and-tool-history), [field variation](known_deviations.md#max-thinking-response-fields) | | TTS error parsing crashes | Tolerate non-JSON errors: [invalid voice](known_deviations.md#invalid-tts-voice-error-shape) | | PCM bytes arrive without format metadata | Configure the format explicitly: [missing headers](known_deviations.md#successful-tts-response-headers) | -| Historical max-vision limitation conflicts with current docs | Check the chosen protocol and current contract: [resolved fixture](known_deviations.md#direct-ecnu-max-image-input) | +| Old max-vision success conflicts with current docs | Current max is text-only; the [September 12 fixture](known_deviations.md#direct-ecnu-max-image-input) predates the September 18 contract change | | `422` | Inspect `detail` and the relevant [request contract](api_reference.md); do not retry the unchanged request | | `429` | Check [current credits/quota](models.md); stop parallel retries | | Timeout or dropped connection after POST | Completion and debit may be unknown; do not automatically resubmit | @@ -44,6 +44,9 @@ disables POST retries, reserves estimated credits, and redacts reports. Its 50-credit default is a cap, not authorization; use the lower approved cap and obtain separate authorization before exceeding 50. Estimates are not actual debit. Keep the same cumulative allowance across reruns, not a new allowance per process. +Token estimates use base/off-peak rates, not the current peak/holiday multiplier; +include that multiplier when checking the approved allowance. Fixed-price +embedding, rerank, image and TTS calls do not use peak pricing. To inspect options without a network request: @@ -65,6 +68,12 @@ and `billable` group other probes. Use `--case` to narrow them. Do not use `all` for an ordinary integration check. Image/TTS probes need authorization covering those billable operations and must not be scheduled automatically. +The runner does not yet have live cases for VL retrieval or image editing. +For those tasks, adapt the [minimal recipes](examples.md) to one authorized, +serial check with synthetic or approved media; preserve the same timeout, +no-retry and redaction boundaries. Offline recipe tests validate request +serialization and error handling, not service availability or output quality. + ## Interpret the evidence | Runner result | Meaning | diff --git a/scripts/smoke_test.py b/scripts/smoke_test.py index 420c327..20ae222 100755 --- a/scripts/smoke_test.py +++ b/scripts/smoke_test.py @@ -100,7 +100,9 @@ "ecnu-plus", "ecnu-max", "ecnu-embedding-small", + "ecnu-embedding-vl", "ecnu-rerank", + "ecnu-rerank-vl", "ecnu-image", "ecnu-tts", } @@ -1302,19 +1304,21 @@ def _minimum_output_credit(model: str | None, payload: Mapping[str, Any] | None) def estimate_consumed_credits( spec: CaseSpec, status: int | None, shape: Mapping[str, Any] ) -> tuple[float, str]: - """Estimate credits conservatively from fixed prices or numeric usage.""" + """Estimate credits from fixed prices or base/off-peak numeric usage.""" if spec.method != "POST" or "MockTransport" in spec.protocol: return 0.0, "no billable ECNU POST" + if spec.model in {"ecnu-embedding-vl", "ecnu-rerank-vl"}: + return 0.2, f"official fixed {spec.model} price per attempted call" if spec.model == "ecnu-embedding-small" or spec.endpoint.endswith("/embeddings"): return 0.05, "official fixed embedding price per attempted call" if spec.endpoint.endswith("/rerank"): - return 0.1, "official fixed rerank price per attempted call" + return 0.05, "official fixed rerank price per attempted call" if spec.endpoint.endswith("/audio/speech"): return 5.0, "official fixed TTS price per attempted call" - if spec.endpoint.endswith("/images/generations"): + if spec.endpoint.endswith(("/images/generations", "/images/edits")): if status is None or status == 200: return 30.0, "official image price; ambiguous attempts counted conservatively" - return 0.0, "definite failed image response is not counted as a successful generation" + return 0.0, "definite failed image response is not counted as a successful image result" counters = shape.get("usage_counters") if not isinstance(counters, dict) or not counters: @@ -1335,7 +1339,7 @@ def estimate_consumed_credits( + (cached_subset + separate_cache) * hit_rate + output_tokens * output_rate ) / 1_000_000 - return credits, f"official {model} miss/hit/output token formula" + return credits, f"{model} base/off-peak token estimate; peak multiplier not applied" def case_response_matches( @@ -2152,7 +2156,7 @@ def _embedding_rerank_cases() -> list[CaseSpec]: ] endpoint = OPENAI_BASE + "/rerank" for case_id, payload, statuses, expectation in rerank_rows: - cases.append(_case(case_id, ("core",), "Cohere-compatible rerank", endpoint, "ecnu-rerank", payload, expectation, "rerank", statuses, cost=0.1)) + cases.append(_case(case_id, ("core",), "Cohere-compatible rerank", endpoint, "ecnu-rerank", payload, expectation, "rerank", statuses, cost=0.05)) return cases @@ -2172,22 +2176,21 @@ def _vision_structured_error_cases() -> list[CaseSpec]: ], "max_tokens": 128, } - for model in ("ecnu-plus", "ecnu-max"): - cases.append( - _case( - "vision_direct_" + model.replace("-", "_"), - ("core",), - "OpenAI-compatible", - endpoint, - model, - {**representative_vision, "model": model}, - f"{model} accepts structured text and image_url data parts.", - "vision", - (200,), - cost=0.1 if model == "ecnu-plus" else 0.25, - payload_factory=lambda context, selected=model: _vision_payload(context, selected), - ) + cases.append( + _case( + "vision_direct_ecnu_plus", + ("core",), + "OpenAI-compatible", + endpoint, + "ecnu-plus", + representative_vision, + "ecnu-plus accepts structured text and image_url data parts.", + "vision", + (200,), + cost=0.1, + payload_factory=lambda context: _vision_payload(context, "ecnu-plus"), ) + ) for model in ("ecnu-plus", "ecnu-max"): for format_type in ("json_schema", "json_object"): response_format: dict[str, Any] = {"type": format_type} diff --git a/tests/test_repository_contracts.py b/tests/test_repository_contracts.py index 8ba51d9..1d76e8b 100644 --- a/tests/test_repository_contracts.py +++ b/tests/test_repository_contracts.py @@ -1,12 +1,17 @@ from __future__ import annotations +import base64 import json +import os import re import shutil import sys import tempfile import unittest +from email import policy +from email.parser import BytesParser from pathlib import Path +from unittest.mock import patch ROOT = Path(__file__).resolve().parents[1] sys.path.insert(0, str(ROOT / "scripts")) @@ -100,6 +105,100 @@ def test_known_deviation_schema_and_status(self) -> None: class CurrentRepositoryContractsTest(unittest.TestCase): + def test_image_edit_recipe_multipart_and_output_validation(self) -> None: + import requests + + text = (ROOT / "references/examples.md").read_text(encoding="utf-8") + section = text.split("## Image generation and editing\n", 1)[1].split("\n## ", 1)[0] + code, = re.findall(r"```python\n(.*?)\n```", section, re.S) + image_bytes = b"synthetic source image" + edited_bytes = b"synthetic edited image" + requests_seen = [] + bodies = [ + {"data": [{"b64_json": base64.b64encode(edited_bytes).decode()}]}, + {"err_message": "rejected", "data": [{"revised_prompt": "****"}]}, + {"data": [{"revised_prompt": "instruction only"}]}, + {"data": [{"b64_json": "not base64!"}]}, + {"data": []}, + ] + + def send(request, **kwargs): + requests_seen.append(request) + self.assertEqual(kwargs["timeout"], 120) + response = requests.Response() + response.status_code = 200 + response._content = json.dumps(body).encode() + return response + + with tempfile.TemporaryDirectory() as directory: + source = Path(directory) / "source.png" + source.write_bytes(image_bytes) + with patch.dict(os.environ, {"ECNU_API_KEY": "test-key"}), \ + patch("requests.sessions.Session.send", side_effect=send): + for index, body in enumerate(bodies): + with self.subTest(body=index): + namespace = {"image_path": source, "prompt": "Keep the subject; change the sky"} + if index == 0: + exec(code, namespace) + self.assertEqual(namespace["edited_bytes"], edited_bytes) + else: + with self.assertRaises((RuntimeError, ValueError)): + exec(code, namespace) + self.assertEqual(source.read_bytes(), image_bytes) + self.assertEqual(len(requests_seen), index + 1) + self.assertTrue(namespace["source"].closed) + + for prompt in ("", " ", "x" * 1025, None): + with self.subTest(prompt_length=len(prompt) if prompt else 0): + with self.assertRaises(ValueError): + exec(code, {"image_path": source, "prompt": prompt}) + self.assertEqual(len(requests_seen), len(bodies)) + + request = requests_seen[0] + self.assertEqual(request.method, "POST") + self.assertEqual(request.url, "https://chat.ecnu.edu.cn/open/api/v1/images/edits") + self.assertEqual(request.headers["Authorization"], "Bearer test-key") + multipart = BytesParser(policy=policy.default).parsebytes( + ("Content-Type: " + request.headers["Content-Type"] + "\r\n\r\n").encode() + + request.body + ) + fields = {part.get_param("name", header="content-disposition"): part + for part in multipart.iter_parts()} + self.assertEqual(set(fields), {"model", "prompt", "response_format", "image"}) + self.assertEqual(fields["model"].get_payload(decode=True), b"ecnu-image") + self.assertEqual(fields["prompt"].get_payload(decode=True), b"Keep the subject; change the sky") + self.assertEqual(fields["response_format"].get_payload(decode=True), b"b64_json") + self.assertEqual(fields["image"].get_payload(decode=True), image_bytes) + + def test_multimodal_recipe_preserves_image_objects(self) -> None: + import requests + + text = (ROOT / "references/examples.md").read_text(encoding="utf-8") + section = text.split("## Multimodal retrieval\n", 1)[1].split("\n## ", 1)[0] + code, = re.findall(r"```python\n(.*?)\n```", section, re.S) + data_url = "data:image/png;base64," + base64.b64encode(b"synthetic image").decode() + namespace = {"image_data_url": data_url} + exec(code, namespace) + for name, endpoint, model in ( + ("embedding_payload", "/embeddings", "ecnu-embedding-vl"), + ("rerank_payload", "/rerank", "ecnu-rerank-vl"), + ): + with self.subTest(endpoint=endpoint): + request = requests.Request( + "POST", "https://example.test" + endpoint, json=namespace[name], + ).prepare() + body = json.loads(request.body) + self.assertEqual(body["model"], model) + items = body["input"] if name == "embedding_payload" else body["documents"] + self.assertEqual(items[0], {"text": "A red flower beside a green leaf", "image": data_url}) + self.assertIsInstance(items[1], str) + if name == "embedding_payload": + self.assertEqual(body["dimensions"], 1024) + self.assertEqual(body["encoding_format"], "float") + else: + self.assertFalse(body["return_documents"]) + self.assertEqual(body["top_n"], len(items)) + def test_documentation_can_quote_embedding_dimensions(self) -> None: with tempfile.TemporaryDirectory() as directory: root = Path(directory) / "ecnu-api" diff --git a/tests/test_smoke_test.py b/tests/test_smoke_test.py index 3e7177e..531ec2a 100644 --- a/tests/test_smoke_test.py +++ b/tests/test_smoke_test.py @@ -521,6 +521,15 @@ def test_models_valid_rejects_empty_ids(self) -> None: ) ) + def test_multimodal_models_are_documented(self) -> None: + models = ["ecnu-embedding-vl", "ecnu-rerank-vl"] + classification = smoke_test.classify_models(models) + self.assertEqual(classification["documented-and-visible"], models) + self.assertEqual(classification["visible-but-undocumented"], []) + self.assertTrue( + set(models) <= set(smoke_test.classify_models([])["documented-but-not-visible"]) + ) + def test_authentication_requires_a_successful_protected_request(self) -> None: discovery = {"object": "list", "data": [{"id": "ecnu-plus"}]} chat = { @@ -714,7 +723,7 @@ def test_image_integrity_requires_complete_valid_png(self) -> None: ) self.assertIsNone(smoke_test._image_dimensions(forged)) - def test_usage_counters_and_official_credit_formula_are_retained(self) -> None: + def test_usage_counters_and_base_credit_estimate_are_retained(self) -> None: spec = self.case("chat_basic_ecnu_max") shape = { "usage_counters": { @@ -726,6 +735,8 @@ def test_usage_counters_and_official_credit_formula_are_retained(self) -> None: credits, basis = smoke_test.estimate_consumed_credits(spec, 200, shape) self.assertAlmostEqual(credits, 0.0372) self.assertIn("ecnu-max", basis) + self.assertIn("base/off-peak", basis) + self.assertIn("peak multiplier not applied", basis) sdk_embedding = self.case("openai_sdk_embedding") fixed, fixed_basis = smoke_test.estimate_consumed_credits( @@ -747,6 +758,33 @@ def test_usage_counters_and_official_credit_formula_are_retained(self) -> None: summary = smoke_test.summarize_response("chat", response) self.assertEqual(summary["usage_counters"], shape["usage_counters"]) + def test_embedding_and_rerank_prices_distinguish_multimodal_models(self) -> None: + for case_id, model, expected in ( + ("embedding_scalar", "ecnu-embedding-small", 0.05), + ("embedding_scalar", "ecnu-embedding-vl", 0.2), + ("rerank_default", "ecnu-rerank", 0.05), + ("rerank_default", "ecnu-rerank-vl", 0.2), + ): + spec = replace(self.case(case_id), model=model) + for status in (200, None): + with self.subTest(model=model, status=status): + credits, _ = smoke_test.estimate_consumed_credits(spec, status, {}) + self.assertEqual(credits, expected) + for spec in smoke_test.build_cases(): + if spec.model == "ecnu-rerank": + self.assertEqual(spec.estimated_credits, 0.05) + + def test_image_edit_and_generation_charge_only_success_or_ambiguous_attempt(self) -> None: + for endpoint in ("/images/generations", "/images/edits"): + spec = replace( + self.case("image_generation_documented"), + endpoint=smoke_test.OPENAI_BASE + endpoint, + ) + for status, expected in ((200, 30.0), (None, 30.0), (400, 0.0), (500, 0.0)): + with self.subTest(endpoint=endpoint, status=status): + credits, _ = smoke_test.estimate_consumed_credits(spec, status, {}) + self.assertEqual(credits, expected) + def test_compatibility_vision_is_unverified_and_behavior_checked(self) -> None: for case_id in ( "responses_max_vision_compatibility", @@ -881,42 +919,54 @@ def test_rerank_requires_nonempty_indexed_numeric_results(self) -> None: ): self.assertFalse(smoke_test.case_response_matches(spec, 200, invalid)) - def test_direct_vision_requires_image_understanding_for_both_models(self) -> None: - for model in ("ecnu-plus", "ecnu-max"): - spec = self.case("vision_direct_" + model.replace("-", "_")) - for status, content, finish_reason, expected in ( - (200, "A red square.", "stop", "pass"), - (200, "A red square.", "length", "mismatch"), - (200, "A red square.", "content_filter", "mismatch"), - (200, "A red square.", None, "mismatch"), - (200, "A red square.", ["stop"], "mismatch"), - (200, "A red square.", {"reason": "stop"}, "mismatch"), - (200, "I cannot inspect images.", "stop", "mismatch"), - (200, "", "stop", "mismatch"), - (422, "Images are unsupported.", None, "mismatch"), - ): - with self.subTest(model=model, status=status, content=content): - payload = ( - {"choices": [{"message": {"content": content}, "finish_reason": finish_reason}]} - if status == 200 else {"detail": content} - ) - response = smoke_test.HttpResult( - status, - {"content-type": "application/json"}, - json.dumps(payload).encode(), + def test_direct_vision_requires_image_understanding(self) -> None: + spec = self.case("vision_direct_ecnu_plus") + for status, content, finish_reason, expected in ( + (200, "A red square.", "stop", "pass"), + (200, "A red square.", "length", "mismatch"), + (200, "A red square.", "content_filter", "mismatch"), + (200, "A red square.", None, "mismatch"), + (200, "A red square.", ["stop"], "mismatch"), + (200, "A red square.", {"reason": "stop"}, "mismatch"), + (200, "I cannot inspect images.", "stop", "mismatch"), + (200, "", "stop", "mismatch"), + (422, "Images are unsupported.", None, "mismatch"), + ): + with self.subTest(status=status, content=content): + payload = ( + {"choices": [{"message": {"content": content}, "finish_reason": finish_reason}]} + if status == 200 else {"detail": content} + ) + response = smoke_test.HttpResult( + status, + {"content-type": "application/json"}, + json.dumps(payload).encode(), + ) + with smoke_test.temporary_artifacts() as directory: + context = smoke_test.RunContext( + "test-key", 1.0, directory, smoke_test.CreditBudget(1.0) ) - with smoke_test.temporary_artifacts() as directory: - context = smoke_test.RunContext( - "test-key", 1.0, directory, smoke_test.CreditBudget(1.0) - ) - with patch.object( - smoke_test, "raw_executor", - return_value=smoke_test.Execution(response, "mock"), - ): - record = smoke_test.run_one(context, spec) - self.assertEqual(record["result"], expected) - if finish_reason == "length": - self.assertEqual(record["actual_response_shape"]["vision_behavior"], "truncated") + with patch.object( + smoke_test, "raw_executor", + return_value=smoke_test.Execution(response, "mock"), + ): + record = smoke_test.run_one(context, spec) + self.assertEqual(record["result"], expected) + if finish_reason == "length": + self.assertEqual(record["actual_response_shape"]["vision_behavior"], "truncated") + + def test_core_vision_uses_plus_and_max_vision_stays_compatibility(self) -> None: + core_vision = [ + case.model for case in smoke_test.build_cases() + if "core" in case.profiles and "vision" in case.response_kind + ] + self.assertEqual(core_vision, ["ecnu-plus"]) + for case in smoke_test.build_cases(): + if "core" in case.profiles and case.model == "ecnu-max": + self.assertNotIn("image_url", json.dumps(case.request_shape), case.case_id) + self.assertNotIn("media_type", json.dumps(case.request_shape), case.case_id) + for case_id in ("responses_max_vision_compatibility", "anthropic_max_vision_compatibility"): + self.assertEqual(self.case(case_id).profiles, {"compatibility"}) class StructuredOutputTest(unittest.TestCase): @@ -1083,7 +1133,7 @@ def test_matrix_contains_high_value_cases(self) -> None: "embedding_token_ids", "embedding_8193_chars", "rerank_top_n_over_count", - "vision_direct_ecnu_max", + "vision_direct_ecnu_plus", "structured_output_ecnu_plus", "structured_output_ecnu_max", "structured_output_json_object_ecnu_plus",