diff --git a/docs/jev/synthetic-pilot-2026-09-29.md b/docs/jev/synthetic-pilot-2026-09-29.md new file mode 100644 index 0000000..164a1cf --- /dev/null +++ b/docs/jev/synthetic-pilot-2026-09-29.md @@ -0,0 +1,40 @@ +# Synthetic Jev pilot review, 2026-09-29 + +The audited synthetic run samples all 36 domains and 24 skills without targeting +evaluation labels. A deterministic 4,000-spec sample from the revised config has +5,944 questions: 45.6% choice, 39.2% noul, and 15.3% score. Difficulty, +ambiguity, and distractor weights favor clearer examples while retaining every +level. The published Tasksource catalog separately includes AG News, emotion, +Banking77, and SuperGLUE CB source tasks; this synthetic run is broad decision +practice, not a replacement for those task families. Hold out direct source +families when measuring transfer to their benchmarks. + +Live pilot artifact: `.synthetic_runs/wide_simple_prompt_pilot_24/` (ignored by +Git). Albert generated 24 states; 23 passed deterministic validation and 19 +passed the stricter critic. The retained 19 states contain 22 real Jev 1.13 +decisions across 15 domains and 15 skills. The critic used Albert DeepSeek +with a stricter prompt, while Jev labeling and the independent GPT-4.1-mini +audit used OpenRouter. OpenRouter usage is subject to the account's pricing and +limits. All 22 audit answers were complete; +the independent auditor disagreed on three and flagged one confident +disagreement. The flagged case concerned three inverter units. The state +explicitly defines a clock offset over five minutes as a mismatch, so Jev's +answer including Unit 12 appears correct and the auditor's omission appears +wrong. Audit flags are diagnostic, not ground truth. + +The critic rejected a choice question whose SLA wording left both a preliminary +report and a final report defensible, and a deployment question where waiting +and inspecting logs were both valid actions. Two retained choice disagreements +still involve plausible alternative actions or ownership, so manual review +remains necessary before claiming clean supervision. The default export keeps +audited disagreements and preserves provenance; use the audit fields for +curation rather than treating the critic or auditor as an oracle. + +The Albert provider accepts an optional `ALBERT_API_KEY_2` environment variable +in the audited and 4,000-state configs. When present, generation and critic +requests rotate across distinct keys; the raw generation records store only a +zero-based credential slot, never a secret. The second key passed model +preflight. A 32-state generation comparison took 69 seconds with two keys +(16 requests per slot) and 89 seconds with one key. Neither short run showed +a rate-limit error, so this suggests a speed benefit but does not establish +independent sustained quotas. diff --git a/src/tasksource/jev/synthetic/config.py b/src/tasksource/jev/synthetic/config.py index 3e12771..2bb2325 100644 --- a/src/tasksource/jev/synthetic/config.py +++ b/src/tasksource/jev/synthetic/config.py @@ -11,6 +11,9 @@ class ProviderConfig: name: str = "albert" api_key_env: str = "ALBERT_API_KEY" + # Extra environment variables are used only when set. Secrets never enter + # configs or manifests; requests rotate across distinct supplied keys. + optional_api_key_envs: list[str] = field(default_factory=list) base_url: str = "https://albert.api.etalab.gouv.fr/v1" model: str = "DeepSeek-V4-Flash" @@ -33,12 +36,31 @@ class SamplerConfig: question_formats: dict = field(default_factory=lambda: {"choice": 0.40, "noul": 0.30, "score": 0.30}) probability_mixed_formats: float = 0.85 probability_all_formats_if_n_ge_3: float = 0.70 + # None preserves the original uniform sampler for existing runs. + difficulty_weights: dict | None = None + ambiguity_weights: dict | None = None + distractor_weights: dict | None = None def __post_init__(self): from .schemas import FORMATS unknown = set(self.question_formats) - set(FORMATS) if unknown or not self.question_formats: raise ValueError(f"question_formats keys must be among {FORMATS}: {sorted(self.question_formats)}") + for name in ("difficulty_weights", "ambiguity_weights", "distractor_weights"): + weights = getattr(self, name) + if weights is not None and (not weights or any(float(v) < 0 for v in weights.values()) + or sum(float(v) for v in weights.values()) <= 0): + raise ValueError(f"{name} must have non-negative weights with positive total mass") + from .specs import AMBIGUITY_LEVELS, DIFFICULTIES + allowed = {"difficulty_weights": DIFFICULTIES, + "ambiguity_weights": AMBIGUITY_LEVELS, + "distractor_weights": range(4)} + for name, values in allowed.items(): + weights = getattr(self, name) + if weights is not None: + unknown = set(map(str, weights)) - set(map(str, values)) + if unknown: + raise ValueError(f"{name} has unknown values: {sorted(unknown)}") @dataclass diff --git a/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev_audited.yaml b/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev_audited.yaml index 7e8be65..43f1ac4 100644 --- a/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev_audited.yaml +++ b/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev_audited.yaml @@ -8,12 +8,13 @@ output_dir: .synthetic_runs provider: name: albert api_key_env: ALBERT_API_KEY + optional_api_key_envs: [ALBERT_API_KEY_2] base_url: https://albert.api.etalab.gouv.fr/v1 model: deepseek-v4-flash-0731 generation: temperature: 0.8 - concurrency: 10 + concurrency: 20 max_output_tokens: 4000 seed: 42 prompt_version: generate_v2 @@ -22,16 +23,21 @@ generation: sampler: seed: 42 n_states: 4000 - questions_per_state: {'1': 0.6, '2': 0.25, '3': 0.1, '4': 0.05} - question_formats: {choice: 0.4, noul: 0.3, score: 0.3} + questions_per_state: {'1': 0.65, '2': 0.25, '3': 0.07, '4': 0.03} + question_formats: {choice: 0.5, noul: 0.4, score: 0.1} probability_mixed_formats: 0.85 - probability_all_formats_if_n_ge_3: 0.7 + probability_all_formats_if_n_ge_3: 0.3 + # Favor clear cases while retaining hard, ambiguous, and noisy examples. + difficulty_weights: {'1': 0.30, '2': 0.25, '3': 0.20, '4': 0.15, '5': 0.10} + ambiguity_weights: {minimal: 0.30, low: 0.25, moderate: 0.20, high: 0.15, extreme: 0.10} + distractor_weights: {'0': 0.35, '1': 0.30, '2': 0.20, '3': 0.15} critic: enabled: true provider: name: albert api_key_env: ALBERT_API_KEY + optional_api_key_envs: [ALBERT_API_KEY_2] base_url: https://albert.api.etalab.gouv.fr/v1 model: deepseek-v4-flash-0731 model: deepseek-v4-flash-0731 diff --git a/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev_night4000.yaml b/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev_night4000.yaml index 1dd90d9..36db3b2 100644 --- a/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev_night4000.yaml +++ b/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev_night4000.yaml @@ -1,14 +1,15 @@ -# 4,000-state overnight run; caches reused from the 1,000-state pilot. +# 4,000-state broad run with a mix of clear and ambiguous decisions. run_name: deepseek_v4_flash_night4000 output_dir: .synthetic_runs provider: name: albert api_key_env: ALBERT_API_KEY + optional_api_key_envs: [ALBERT_API_KEY_2] base_url: https://albert.api.etalab.gouv.fr/v1 model: deepseek-v4-flash-0731 generation: temperature: 0.8 - concurrency: 10 + concurrency: 20 max_output_tokens: 4000 seed: 42 prompt_version: generate_v2 @@ -17,21 +18,25 @@ sampler: seed: 42 n_states: 4000 questions_per_state: - '1': 0.6 + '1': 0.65 '2': 0.25 - '3': 0.1 - '4': 0.05 + '3': 0.07 + '4': 0.03 question_formats: - choice: 0.4 - noul: 0.3 - score: 0.3 + choice: 0.5 + noul: 0.4 + score: 0.1 probability_mixed_formats: 0.85 - probability_all_formats_if_n_ge_3: 0.7 + probability_all_formats_if_n_ge_3: 0.3 + difficulty_weights: {'1': 0.30, '2': 0.25, '3': 0.20, '4': 0.15, '5': 0.10} + ambiguity_weights: {minimal: 0.30, low: 0.25, moderate: 0.20, high: 0.15, extreme: 0.10} + distractor_weights: {'0': 0.35, '1': 0.30, '2': 0.20, '3': 0.15} critic: enabled: true provider: name: albert api_key_env: ALBERT_API_KEY + optional_api_key_envs: [ALBERT_API_KEY_2] base_url: https://albert.api.etalab.gouv.fr/v1 model: deepseek-v4-flash-0731 model: deepseek-v4-flash-0731 diff --git a/src/tasksource/jev/synthetic/critic.py b/src/tasksource/jev/synthetic/critic.py index 511961f..37dc499 100644 --- a/src/tasksource/jev/synthetic/critic.py +++ b/src/tasksource/jev/synthetic/critic.py @@ -29,7 +29,8 @@ def load_critic_prompt(version: str) -> str: def critic_cache_key(model: str, temperature: float, prompt: str, bundle: dict, endpoint: str = "") -> str: """``endpoint`` names the provider (``name@base_url``): one model name can be served by several.""" canonical = json.dumps( - {"state_id": bundle.get("state_id"), "state": bundle.get("state"), + {"state_id": bundle.get("state_id"), "domain": bundle.get("domain"), + "state": bundle.get("state"), "questions": bundle.get("questions")}, sort_keys=True, ensure_ascii=False) return hashlib.sha256( @@ -76,7 +77,8 @@ async def _critique_one(sem, client, model: str, temperature: float, "provider": "mock", "model": model, "verdict": verdict}, ensure_ascii=False, indent=2), encoding="utf-8") return {"state_id": bundle["state_id"], **verdict} - payload = {"state_id": bundle["state_id"], "state": bundle["state"], + payload = {"state_id": bundle["state_id"], "domain": bundle.get("domain"), + "state": bundle["state"], "questions": bundle["questions"]} prompt = template.replace("{{BUNDLE_JSON}}", json.dumps(payload, ensure_ascii=False, indent=2)) async with sem: @@ -108,22 +110,26 @@ async def critique_bundles_async(cfg, bundles: list[dict], raw_dir: Path) -> lis return [{"state_id": b["state_id"], "pass": True, "issues": [], "score": 1.0} for b in bundles] template = load_critic_prompt(cfg.critic.prompt_version) provider = cfg.critic_provider() - client = None + clients = [None] if provider.name != "mock": - client = providers.make_client(provider, providers.require_api_key(provider)) + clients = [providers.make_client(provider, key) + for key in providers.available_api_keys(provider)] try: sem = asyncio.Semaphore(max(1, cfg.generation.concurrency)) - pacer = RequestPacer(cfg.critic.requests_per_minute) if client is not None else None - out = await asyncio.gather(*[_critique_one(sem, client, cfg.critic.model, + pacers = [RequestPacer(cfg.critic.requests_per_minute) if client is not None + else None for client in clients] + out = await asyncio.gather(*[_critique_one(sem, clients[i % len(clients)], cfg.critic.model, cfg.critic.temperature, template, b, raw_dir, - pacer, f"{provider.name}@{provider.base_url.rstrip('/')}") - for b in bundles]) + pacers[i % len(clients)], + f"{provider.name}@{provider.base_url.rstrip('/')}") + for i, b in enumerate(bundles)]) finally: - if client is not None: - try: - await client.close() - except Exception: - pass + for client in clients: + if client is not None: + try: + await client.close() + except Exception: + pass return list(out) diff --git a/src/tasksource/jev/synthetic/generate.py b/src/tasksource/jev/synthetic/generate.py index 125f406..4aeffbd 100644 --- a/src/tasksource/jev/synthetic/generate.py +++ b/src/tasksource/jev/synthetic/generate.py @@ -146,7 +146,8 @@ def parse_bundle(text: str, spec: dict) -> dict: async def _generate_one(sem: asyncio.Semaphore, client, cfg, prompt_template: str, - p_hash: str, c_hash: str, spec: dict, raw_dir: Path) -> dict: + p_hash: str, c_hash: str, spec: dict, raw_dir: Path, + credential_slot: int = 0) -> dict: req_hash = providers.request_hash(c_hash, p_hash, spec) cached = raw_dir / f"{req_hash}.json" if cached.exists(): @@ -159,6 +160,7 @@ async def _generate_one(sem: asyncio.Semaphore, client, cfg, prompt_template: st record = {"request_hash": req_hash, "spec": spec, "prompt": prompt, "provider": "mock", "requested_model": cfg.provider.model, "returned_model": cfg.provider.model, + "credential_slot": credential_slot, "temperature": cfg.generation.temperature, "prompt_version": cfg.generation.prompt_version, "bundle": bundle} cached.write_text(json.dumps(record, ensure_ascii=False, indent=2), encoding="utf-8") @@ -187,6 +189,7 @@ async def _generate_one(sem: asyncio.Semaphore, client, cfg, prompt_template: st "raw_response": result["raw"], "raw_text": result["text"], "provider": cfg.provider.name, "requested_model": cfg.provider.model, "returned_model": result["returned_model"], + "credential_slot": credential_slot, "temperature": cfg.generation.temperature, "prompt_version": cfg.generation.prompt_version, "bundle": bundle} cached.write_text(json.dumps(record, ensure_ascii=False, indent=2), encoding="utf-8") @@ -194,11 +197,11 @@ async def _generate_one(sem: asyncio.Semaphore, client, cfg, prompt_template: st async def _safe_generate_one(sem, client, cfg, prompt_template, p_hash, c_hash, - spec, raw_dir) -> dict: + spec, raw_dir, credential_slot: int = 0) -> dict: """Fault isolation: one bad spec must not kill a 1k run.""" try: return await _generate_one(sem, client, cfg, prompt_template, p_hash, - c_hash, spec, raw_dir) + c_hash, spec, raw_dir, credential_slot) except Exception as exc: # noqa: BLE001 return {"spec": spec, "bundle": None, "request_hash": None, "cached": False, "record": None, "error": str(exc)} @@ -209,16 +212,26 @@ async def generate_bundles_async(cfg, specs: list[dict], raw_dir: Path) -> list[ prompt_template = load_prompt(cfg.generation.prompt_version) p_hash = prompt_hash(prompt_template) c_hash = generation_cache_key(cfg) - client = None + clients = [None] if cfg.provider.name != "mock": - api_key = providers.require_api_key(cfg.provider) - client = providers.make_client(cfg.provider, api_key) + clients = [providers.make_client(cfg.provider, key) + for key in providers.available_api_keys(cfg.provider)] sem = asyncio.Semaphore(max(1, cfg.generation.concurrency)) - tasks = [_safe_generate_one(sem, client, cfg, prompt_template, p_hash, c_hash, spec, raw_dir) - for spec in specs] - results = [] - for coro in asyncio.as_completed(tasks): - results.append(await coro) + tasks = [_safe_generate_one(sem, clients[i % len(clients)], cfg, + prompt_template, p_hash, c_hash, spec, raw_dir, + i % len(clients)) + for i, spec in enumerate(specs)] + try: + results = [] + for coro in asyncio.as_completed(tasks): + results.append(await coro) + finally: + for client in clients: + if client is not None: + try: + await client.close() + except Exception: + pass errors = [r for r in results if r.get("bundle") is None] if errors: (raw_dir / "errors.jsonl").write_text( diff --git a/src/tasksource/jev/synthetic/prompts/critic_v1.txt b/src/tasksource/jev/synthetic/prompts/critic_v1.txt index 4592d23..753ed45 100644 --- a/src/tasksource/jev/synthetic/prompts/critic_v1.txt +++ b/src/tasksource/jev/synthetic/prompts/critic_v1.txt @@ -8,5 +8,15 @@ Check: 2. Questions are not paraphrases of each other; each tests a distinct aspect/skill. 3. Questions are not mutually inconsistent and do not leak the answer. 4. Requested distinct formats/skills are actually represented. +5. Each yes/no question asks about a single proposition, and each score question + matches the meaning of its ordered rubric. Reject mismatches such as urgency + with likelihood levels or a numeric scale without defined endpoints. +6. The state belongs to its stated domain and does not mention Jev or the + dataset construction process. +7. Reject a question when the state omits the policy, deadline, current time, + or decision standard needed to answer it. Reject specialist safety or legal + decisions that require outside rules not supplied in the state. +8. For each choice question, reject if two or more options are defensible from + the state or if the best option depends on an unstated interpretation. Output STRICT JSON only: {"pass": bool, "issues": [str], "score": float}. diff --git a/src/tasksource/jev/synthetic/prompts/generate_v2.txt b/src/tasksource/jev/synthetic/prompts/generate_v2.txt index 04085b3..1f83b81 100644 --- a/src/tasksource/jev/synthetic/prompts/generate_v2.txt +++ b/src/tasksource/jev/synthetic/prompts/generate_v2.txt @@ -11,14 +11,23 @@ SPEC: Rules: - Generate ONE coherent state plus EXACTLY the requested number of questions, in order. - Every question must be answerable or meaningfully assessable from the SAME state. +- If a decision depends on a policy, threshold, deadline, current time, or + authority, include that information explicitly in the state. Do not require + outside legal, medical, aviation, or other specialist rules to pick an answer. - Questions must test DISTINCT aspects of the state (not paraphrases). - For choice questions output EXACTLY n_options options (use the requested count). +- Make exactly ONE choice option best supported by the state. Other options may + be plausible, but the state must contain evidence that rules them out. Avoid + pairs of options that are both valid next steps under the stated policy. - For score questions the spec gives ordered "criteria": reuse them VERBATIM, in order, as the "options" array, with matching "min" (first index) and - "max" (last index). Never rename, merge, or reorder criteria. -- For noul questions output only the question text. + "max" (last index). Never rename, merge, or reorder criteria. State what the + scale measures and what its endpoints mean; match the question to the rubric. +- For noul questions output only the question text. Ask about ONE yes/no + proposition, never "positive or negative" or another either/or choice. - Include requested irrelevant/distractor information naturally in the state. - Do NOT state or hint which option is correct; do not make one option lexically obvious. - Do NOT mention difficulty, ambiguity, skill names, evidence, distractors, or any metadata. +- Do NOT mention Jev or a dataset in the state or questions. - Do NOT include phrases like "correct answer" or "the answer is". - Output STRICT JSON only, no markdown fences, with keys: {"state": str, "questions": [{"question": str, "options"?: [str], "min"?: int, "max"?: int}]}. diff --git a/src/tasksource/jev/synthetic/providers.py b/src/tasksource/jev/synthetic/providers.py index d12d425..70872bf 100644 --- a/src/tasksource/jev/synthetic/providers.py +++ b/src/tasksource/jev/synthetic/providers.py @@ -34,6 +34,16 @@ def require_api_key(provider: ProviderConfig) -> str: return api_key +def available_api_keys(provider: ProviderConfig) -> list[str]: + """Primary key plus distinct, present optional keys in configured order.""" + keys = [require_api_key(provider)] + for name in provider.optional_api_key_envs: + value = os.getenv(name, "") + if value and value not in keys: + keys.append(value) + return keys + + def make_client(provider: ProviderConfig, api_key: str): """Build a generic OpenAI-compatible async client.""" try: @@ -51,24 +61,25 @@ async def preflight(provider: ProviderConfig) -> PreflightResult: Must run before sampling specs or writing artifacts (except the run dir). """ - api_key = require_api_key(provider) + api_keys = available_api_keys(provider) if provider.name == "mock": return PreflightResult(provider="mock", model=provider.model, returned_model=provider.model, base_url="mock://") - client = make_client(provider, api_key) - try: - models = await client.models.list() - finally: + for api_key in api_keys: + client = make_client(provider, api_key) try: - await client.close() - except Exception: - pass - available = [m.id for m in models.data] - if provider.model not in available: - raise RuntimeError( - f"Model {provider.model!r} not listed by {provider.name} " - f"(base_url={provider.base_url}). Available: {available[:20]}" - ) + models = await client.models.list() + available = [m.id for m in models.data] + if provider.model not in available: + raise RuntimeError( + f"Model {provider.model!r} not listed by {provider.name} " + f"(base_url={provider.base_url}). Available: {available[:20]}" + ) + finally: + try: + await client.close() + except Exception: + pass returned = provider.model return PreflightResult(provider=provider.name, model=provider.model, returned_model=returned, base_url=provider.base_url) diff --git a/src/tasksource/jev/synthetic/specs.py b/src/tasksource/jev/synthetic/specs.py index 7ff08e1..d73af9f 100644 --- a/src/tasksource/jev/synthetic/specs.py +++ b/src/tasksource/jev/synthetic/specs.py @@ -135,6 +135,25 @@ ["unacceptable", "poor", "adequate", "good", "excellent"], ] +# Ordered words must describe the skill being rated. Unlisted skills use a +# numeric scale, whose endpoints the generator must explain in the question. +SEMANTIC_RUBRICS_BY_SKILL = { + "severity": [SEMANTIC_RUBRICS[0], SEMANTIC_RUBRICS[4]], + "triage_priority": [["routine", "low", "medium", "high", "critical"]], + "sentiment": [SEMANTIC_RUBRICS[1]], + "fraud_likelihood": [SEMANTIC_RUBRICS[5]], + "churn_risk": [SEMANTIC_RUBRICS[5]], + "needs_escalation": [SEMANTIC_RUBRICS[5]], + "urgency": [["not urgent", "low urgency", "moderate urgency", + "high urgency", "immediate"]], + "compliance_risk": [SEMANTIC_RUBRICS[0]], + "data_sensitivity": [["public", "internal", "confidential", "restricted"]], + "customer_effort": [["none", "low", "moderate", "high", "very high"]], + "groundedness": [SEMANTIC_RUBRICS[6]], + "completeness": [SEMANTIC_RUBRICS[2]], + "resolution_confidence": [SEMANTIC_RUBRICS[5]], +} + def domain_skills(domain: str) -> list[str]: """Skill pool compatible with a domain (falls back to all skills).""" @@ -166,6 +185,14 @@ def sample_n_questions(rng: random.Random, dist: dict) -> int: return int(_weighted_choice(rng, {k: float(v) for k, v in dist.items()})) +def _configured_choice(rng: random.Random, values: list, weights: dict | None): + if weights is None: + return rng.choice(values) + choices = [value for value in values if str(value) in weights or value in weights] + masses = [float(weights.get(value, weights.get(str(value), 0))) for value in choices] + return rng.choices(choices, weights=masses, k=1)[0] + + def sample_formats(rng: random.Random, n: int, format_weights: dict, p_mixed: float = 0.85, p_all_if_ge3: float = 0.70) -> list[str]: """Sample n formats honoring coverage + mixed-format preferences.""" @@ -207,11 +234,12 @@ def sample_question_spec(rng: random.Random, fmt: str, skill: str) -> dict: # Score: Tasksource represents these as ORDERED CRITERIA with the # target aligned to them — a healthy mix of numeric scales and # named rubrics. - if rng.random() < 0.5: + rubrics = SEMANTIC_RUBRICS_BY_SKILL.get(skill, []) + if rng.random() < 0.5 or not rubrics: lo, hi = rng.choice(SCORE_RANGES) return {"format": "score", "skill": skill, "min": lo, "max": hi, "criteria": numeric_criteria(lo, hi)} - rubric = list(rng.choice(SEMANTIC_RUBRICS)) + rubric = list(rng.choice(rubrics)) return {"format": "score", "skill": skill, "min": 0, "max": len(rubric) - 1, "criteria": rubric} @@ -231,11 +259,14 @@ def sample_spec(index: int, rng: random.Random, sampler_cfg) -> dict: "domain": domain, "scenario_type": rng.choice(SCENARIO_TYPES), "style": rng.choice(domain_styles(domain)), - "difficulty": rng.choice(DIFFICULTIES), - "ambiguity": rng.choice(AMBIGUITY_LEVELS), + "difficulty": _configured_choice(rng, DIFFICULTIES, + sampler_cfg.difficulty_weights), + "ambiguity": _configured_choice(rng, AMBIGUITY_LEVELS, + sampler_cfg.ambiguity_weights), "state_length": rng.choice(["short", "medium", "long"]), "evidence": rng.choice(EVIDENCE_STRUCTURES), - "distractors": rng.choice([0, 1, 2, 3]), + "distractors": _configured_choice(rng, [0, 1, 2, 3], + sampler_cfg.distractor_weights), "certainty_hint": rng.choice(NOUL_CERTAINTY), "questions": [sample_question_spec(rng, fmt, skill) for fmt, skill in zip(formats, skills)], diff --git a/src/tasksource/jev/synthetic/validate.py b/src/tasksource/jev/synthetic/validate.py index 192a49d..aafb4ed 100644 --- a/src/tasksource/jev/synthetic/validate.py +++ b/src/tasksource/jev/synthetic/validate.py @@ -9,6 +9,8 @@ LEAK_TOKENS = ("difficulty", "ambiguity", "skill", "distractor", "state_length", "evidence", "certainty") ANSWER_LEAK = re.compile(r"correct (answer|option|choice)|answer is\b", re.IGNORECASE) +MODEL_LEAK = re.compile(r"\bjev\b", re.IGNORECASE) +EITHER_OR_LABEL = re.compile(r"\b(?:positive\s+or\s+negative|negative\s+or\s+positive|yes\s+or\s+no|no\s+or\s+yes)\b", re.IGNORECASE) OPEN_QUESTION = re.compile(r"\b(?:what|which|who|where|when)\b", re.IGNORECASE) EVENT_PROBABILITY = re.compile( r"\b(?:likelihood|probability|chance|risk|confidence)\s+that\b|" @@ -17,6 +19,8 @@ def valid_noul_question(text: str) -> bool: """A noul target is a probability for one proposition, not a free-form answer.""" + if EITHER_OR_LABEL.search(text): + return False if EVENT_PROBABILITY.search(text): return True if OPEN_QUESTION.search(text) or re.search(r"\bhow\b", text, re.IGNORECASE): @@ -40,15 +44,24 @@ def validate_bundle(bundle: dict, spec: dict | None = None) -> list[str]: break if ANSWER_LEAK.search(state): errors.append("state leaks correct answer phrasing") + if MODEL_LEAK.search(state): + errors.append("state mentions the target model") texts = [q.get("question", "") for q in bundle.get("questions", [])] for text in texts: if not text or len(text.strip()) < 10: errors.append("empty/truncated question") if ANSWER_LEAK.search(text): errors.append("question leaks correct answer phrasing") + if MODEL_LEAK.search(text): + errors.append("question mentions the target model") for q in bundle.get("questions", []): if q.get("format") == "noul" and not valid_noul_question(q.get("question", "")): errors.append(f"noul {q.get('question_id')} must ask about one yes/no proposition") + if (q.get("format") == "score" + and list(q.get("options", [])) == + ["very unlikely", "unlikely", "possible", "likely", "very likely"] + and re.search(r"\burgen(?:t|cy)\b", q.get("question", ""), re.IGNORECASE)): + errors.append(f"score {q.get('question_id')} uses likelihood levels for urgency") if len(set(texts)) != len(texts): errors.append("duplicate question texts in bundle") options_seen = [tuple(q.get("options", [])) for q in bundle.get("questions", []) if q.get("format") == "choice"] diff --git a/tests/test_jev_synthetic.py b/tests/test_jev_synthetic.py index e13c09b..e9515a0 100644 --- a/tests/test_jev_synthetic.py +++ b/tests/test_jev_synthetic.py @@ -96,6 +96,19 @@ def test_score_specs_have_ordered_criteria(self): and q["criteria"][0].lstrip("-").isdigit()) self.assertGreater(numeric, 0) self.assertGreater(len(score_qs) - numeric, 0) # semantic rubrics too + for q in score_qs: + if not q["criteria"][0].lstrip("-").isdigit(): + self.assertIn(q["criteria"], specs_mod.SEMANTIC_RUBRICS_BY_SKILL[q["skill"]]) + + def test_audited_run_is_broad_with_smaller_score_share(self): + path = (Path(__file__).resolve().parents[1] / "src" / "tasksource" / "jev" + / "synthetic" / "configs" / "albert_deepseek_v4_flash_jev_audited.yaml") + specs = specs_mod.sample_specs(load_config(str(path)).sampler, 4000) + formats = Counter(q["format"] for spec in specs for q in spec["questions"]) + self.assertTrue(0.12 < formats["score"] / sum(formats.values()) < 0.18) + self.assertEqual({spec["domain"] for spec in specs}, set(specs_mod.DOMAINS)) + self.assertEqual({spec["difficulty"] for spec in specs}, set(specs_mod.DIFFICULTIES)) + self.assertEqual({spec["ambiguity"] for spec in specs}, set(specs_mod.AMBIGUITY_LEVELS)) class BundleTest(unittest.TestCase): @@ -117,12 +130,31 @@ def test_noul_requires_probability_for_a_proposition(self): "Which team should own this ticket?", "According to the report, what is the timestamp on the monitor?", "List all of the action items assigned to the agent.", + "Is the customer's sentiment positive or negative?", ] for text in valid: self.assertTrue(validate_mod.valid_noul_question(text), text) for text in invalid: self.assertFalse(validate_mod.valid_noul_question(text), text) + def test_rejects_pilot_format_and_model_leaks(self): + bundle = { + "state_id": "s", "state": "The vendor delayed a shipment and the client called support.", + "questions": [ + {"question_id": "q0", "format": "noul", + "question": "Is the client's sentiment positive or negative?"}, + {"question_id": "q1", "format": "score", + "question": "How urgent is the shipment issue?", + "options": ["very unlikely", "unlikely", "possible", "likely", "very likely"]}, + {"question_id": "q2", "format": "noul", + "question": "Should Jev approve the refund?"}, + ], + } + errors = validate_mod.validate_bundle(bundle) + self.assertTrue(any("one yes/no proposition" in error for error in errors)) + self.assertTrue(any("likelihood levels for urgency" in error for error in errors)) + self.assertTrue(any("target model" in error for error in errors)) + def test_flat_preserves_grouping(self): spec = specs_mod.sample_specs(_cfg().sampler, 5)[0] bundle = mock_realization(spec) @@ -521,6 +553,17 @@ def test_score_criteria_capped_at_ten(self): class PreflightTest(unittest.TestCase): + def test_optional_api_keys_are_distinct_and_need_not_be_set(self): + from unittest.mock import patch + from tasksource.jev.synthetic.config import ProviderConfig + provider = ProviderConfig(name="albert", api_key_env="TEST_PRIMARY", + optional_api_key_envs=["TEST_EXTRA", "TEST_DUPLICATE"]) + with patch.dict("os.environ", {"TEST_PRIMARY": "key-one"}, clear=True): + self.assertEqual(providers.available_api_keys(provider), ["key-one"]) + with patch.dict("os.environ", {"TEST_PRIMARY": "key-one", "TEST_EXTRA": "key-two", + "TEST_DUPLICATE": "key-one"}, clear=True): + self.assertEqual(providers.available_api_keys(provider), ["key-one", "key-two"]) + def test_missing_key_raises(self): import os from tasksource.jev.synthetic.config import ProviderConfig