From f39e7b5639d0622a2f1b325e22ddb0655b9cf2db Mon Sep 17 00:00:00 2001 From: Damien Sileo Date: Tue, 29 Sep 2026 16:20:21 +0200 Subject: [PATCH] Broaden synthetic Jev with interpretive text tasks --- docs/jev/synthetic-pilot-2026-09-29.md | 65 +++++++++++++++++++ src/tasksource/jev/synthetic/config.py | 10 ++- .../configs/albert_deepseek_v4_flash.yaml | 4 +- .../configs/albert_deepseek_v4_flash_jev.yaml | 4 +- .../albert_deepseek_v4_flash_jev_audited.yaml | 9 +-- ...lbert_deepseek_v4_flash_jev_night4000.yaml | 5 +- .../jev/synthetic/configs/mock_pilot.yaml | 4 +- .../jev/synthetic/configs/openai_luna.yaml | 4 +- src/tasksource/jev/synthetic/critic.py | 35 ++++++++-- .../jev/synthetic/prompts/critic.txt | 51 +++++++++++++++ .../jev/synthetic/prompts/critic_v1.txt | 22 ------- .../prompts/{generate_v2.txt => generate.txt} | 25 +++++++ .../jev/synthetic/prompts/generate_v1.txt | 22 ------- ...teacher_audit_v1.txt => teacher_audit.txt} | 0 .../jev/synthetic/samples/README.md | 2 +- src/tasksource/jev/synthetic/specs.py | 46 +++++++++++-- src/tasksource/jev/synthetic/validate.py | 12 ++-- tests/test_jev_synthetic.py | 38 +++++++++++ 18 files changed, 278 insertions(+), 80 deletions(-) create mode 100644 src/tasksource/jev/synthetic/prompts/critic.txt delete mode 100644 src/tasksource/jev/synthetic/prompts/critic_v1.txt rename src/tasksource/jev/synthetic/prompts/{generate_v2.txt => generate.txt} (54%) delete mode 100644 src/tasksource/jev/synthetic/prompts/generate_v1.txt rename src/tasksource/jev/synthetic/prompts/{teacher_audit_v1.txt => teacher_audit.txt} (100%) diff --git a/docs/jev/synthetic-pilot-2026-09-29.md b/docs/jev/synthetic-pilot-2026-09-29.md index 164a1cf..007a0be 100644 --- a/docs/jev/synthetic-pilot-2026-09-29.md +++ b/docs/jev/synthetic-pilot-2026-09-29.md @@ -38,3 +38,68 @@ preflight. A 32-state generation comparison took 69 seconds with two keys (16 requests per slot) and 89 seconds with one key. Neither short run showed a rate-limit error, so this suggests a speed benefit but does not establish independent sustained quotas. + +## Broad interpretive text understanding + +The audited and 4,000-state configurations add low-weight general reading skills +through the ordinary domain sampler. The current 14 skills cover topic, emotion, +communicative intent, claim support, stance, document purpose, main point, +implicit concern, intended audience, argument role, stakeholder perspective, +social implication, evidence strength, and message tone. They share generation, +validation, critic, Jev annotation, and independent audit with other skills. +Arithmetic, chronology, entity lookup, and literal retrieval are left to the +procedural segment. There are no benchmark labels or benchmark-specific paths. +Prompt files have stable names (`generate.txt`, `critic.txt`, and +`teacher_audit.txt`) rather than version suffixes. + +A deterministic 4,000-spec draw yielded 5,890 questions, including 606 general +reading questions (10.3%). All 36 domains appeared. Each of the 14 general +skills appeared in at least 19 domains. The overall mix still includes existing +skills such as toxicity, sentiment, and groundedness. Difficulty levels 1–5 +remain available; the general segment intentionally contains both simple text +classification and questions requiring several cues. + +The first natural live pilot, `.synthetic_runs/natural_general_pilot_115/`, used +an earlier eight-skill candidate mix. It yielded 96 retained states and 123 +Jev decisions from 115 generated states. All 18 retained general questions had +Jev/auditor agreement, but manual review found an entity-type question that +classified an issue rather than an entity. That, plus overlap with the +procedural generators, prompted the current interpretive mix. + +The current natural live pilot is +`.synthetic_runs/interpretive_general_pilot_120/` (ignored by Git). Its 120 +Albert generations produced 117 validated states, 94 critic-approved and +retained states, and 119 Jev decisions. Sixteen retained questions used nine +of the general reading skills. The independent auditor answered 118 of 119 +questions, disagreed with Jev on 12 overall and one general question, flagged +34 questions for review, and found no confident disagreements; 93 of 94 states +passed audit. Manual review of the 16 general questions found useful variety, +but also drift: two `claim_support` questions asked for a main concern, an +`audience_inference` question became ticket routing, an `implicit_concern` +question guessed a customer's feelings without the customer's own words, and a +`practical_implication` question duplicated policy application. The last skill +has been replaced by `social_implication`, and the generation and critic prompts +now explicitly reject those drifts. Jev/auditor agreement did not catch them. + +That current pilot took 582 seconds end to end with one Albert key and pilot +critic/auditor limits of 120 requests per minute: about 13,900 retained states +or 17,700 decisions per day if that rate holds. The earlier 115-state pilot +took 481 seconds with two Albert keys: about 17,200 retained states or 22,100 +decisions per day by the same extrapolation. These are short-run estimates, +not sustained throughput guarantees; the production configs use 40 critic and +auditor requests per minute, and quota, retries, cost, and longer-run quality +may change throughput. Auditor agreement is diagnostic, not ground truth. + +A focused quality probe, `.synthetic_runs/interpretive_skill_quality_probe_16/`, +sampled four specs each for claim support, intended audience, implicit concern, +and social implication from the ordinary 4,000-spec draw. Fourteen of 16 states +passed the original validator and critic, yielding 21 decisions. Manual review +found an ambiguous intended audience, a negotiation implication with several +plausible readings, and a question that projected today's idle crew into a +client visit tomorrow. The auditor flagged the first two but agreed on the +third. The critic now requires a per-question, exact state quote and explicit +skill, support, and unique-answer checks. On the same 15 validated states this +stricter DeepSeek check rejected the ambiguous negotiation question, although +it still missed the unwarranted projection about tomorrow. This is a concrete +remaining quality risk. A larger manual sample is needed before treating an +unattended 4,000-state run as clean training data. diff --git a/src/tasksource/jev/synthetic/config.py b/src/tasksource/jev/synthetic/config.py index 2bb2325..31e7c9a 100644 --- a/src/tasksource/jev/synthetic/config.py +++ b/src/tasksource/jev/synthetic/config.py @@ -24,7 +24,7 @@ class GenerationConfig: concurrency: int = 20 max_output_tokens: int = 4000 seed: int = 42 - prompt_version: str = "generate_v1" + prompt_version: str = "generate" n_states: int = 1000 @@ -40,12 +40,16 @@ class SamplerConfig: difficulty_weights: dict | None = None ambiguity_weights: dict | None = None distractor_weights: dict | None = None + # Optional low-weight general language skills enter the ordinary domain mix. + general_text_skill_weight: float = 0.0 def __post_init__(self): from .schemas import FORMATS unknown = set(self.question_formats) - set(FORMATS) if unknown or not self.question_formats: raise ValueError(f"question_formats keys must be among {FORMATS}: {sorted(self.question_formats)}") + if not math.isfinite(self.general_text_skill_weight) or self.general_text_skill_weight < 0: + raise ValueError("general_text_skill_weight must be finite and non-negative") for name in ("difficulty_weights", "ambiguity_weights", "distractor_weights"): weights = getattr(self, name) if weights is not None and (not weights or any(float(v) < 0 for v in weights.values()) @@ -71,7 +75,7 @@ class CriticConfig: # When `provider` is absent, it inherits the generator provider. provider: ProviderConfig | None = None model: str = "deepseek-v4-flash-0731" - prompt_version: str = "critic_v1" + prompt_version: str = "critic" temperature: float = 0.0 requests_per_minute: int = 40 @@ -101,7 +105,7 @@ class TeacherAuditConfig: # check, explicitly choose a different provider/model from the generator. provider: ProviderConfig | None = None model: str = "deepseek-v4-flash-0731" - prompt_version: str = "teacher_audit_v1" + prompt_version: str = "teacher_audit" temperature: float = 0.0 requests_per_minute: int = 40 min_teacher_confidence: float = 0.85 diff --git a/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash.yaml b/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash.yaml index 19d375e..6863c1b 100644 --- a/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash.yaml +++ b/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash.yaml @@ -15,7 +15,7 @@ generation: concurrency: 10 max_output_tokens: 4000 seed: 42 - prompt_version: generate_v2 + prompt_version: generate n_states: 1000 sampler: @@ -34,7 +34,7 @@ critic: base_url: https://albert.api.etalab.gouv.fr/v1 model: deepseek-v4-flash-0731 model: deepseek-v4-flash-0731 - prompt_version: critic_v1 + prompt_version: critic temperature: 0.0 annotator: diff --git a/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev.yaml b/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev.yaml index a10f9d4..b26d6e7 100644 --- a/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev.yaml +++ b/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev.yaml @@ -13,7 +13,7 @@ generation: concurrency: 10 max_output_tokens: 4000 seed: 42 - prompt_version: generate_v2 + prompt_version: generate n_states: 1000 sampler: @@ -32,7 +32,7 @@ critic: base_url: https://albert.api.etalab.gouv.fr/v1 model: deepseek-v4-flash-0731 model: deepseek-v4-flash-0731 - prompt_version: critic_v1 + prompt_version: critic temperature: 0.0 requests_per_minute: 40 diff --git a/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev_audited.yaml b/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev_audited.yaml index 43f1ac4..1fe0ff8 100644 --- a/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev_audited.yaml +++ b/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev_audited.yaml @@ -1,4 +1,4 @@ -# DeepSeek generation + independent OpenAI-model teacher audit + pinned Jev targets. +# DeepSeek generation and critic + independent teacher audit + pinned Jev targets. # This is intentionally more expensive than the overnight baseline. The auditor # never sees Jev's probabilities; it independently answers each question and only # flags high-confidence disagreements. @@ -17,7 +17,7 @@ generation: concurrency: 20 max_output_tokens: 4000 seed: 42 - prompt_version: generate_v2 + prompt_version: generate n_states: 4000 sampler: @@ -31,6 +31,7 @@ sampler: difficulty_weights: {'1': 0.30, '2': 0.25, '3': 0.20, '4': 0.15, '5': 0.10} ambiguity_weights: {minimal: 0.30, low: 0.25, moderate: 0.20, high: 0.15, extreme: 0.10} distractor_weights: {'0': 0.35, '1': 0.30, '2': 0.20, '3': 0.15} + general_text_skill_weight: 0.3 critic: enabled: true @@ -41,7 +42,7 @@ critic: base_url: https://albert.api.etalab.gouv.fr/v1 model: deepseek-v4-flash-0731 model: deepseek-v4-flash-0731 - prompt_version: critic_v1 + prompt_version: critic temperature: 0.0 requests_per_minute: 40 @@ -61,7 +62,7 @@ teacher_audit: base_url: https://openrouter.ai/api/v1 model: openai/gpt-4.1-mini model: openai/gpt-4.1-mini - prompt_version: teacher_audit_v1 + prompt_version: teacher_audit temperature: 0.0 requests_per_minute: 40 min_teacher_confidence: 0.85 diff --git a/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev_night4000.yaml b/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev_night4000.yaml index 36db3b2..e08728f 100644 --- a/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev_night4000.yaml +++ b/src/tasksource/jev/synthetic/configs/albert_deepseek_v4_flash_jev_night4000.yaml @@ -12,7 +12,7 @@ generation: concurrency: 20 max_output_tokens: 4000 seed: 42 - prompt_version: generate_v2 + prompt_version: generate n_states: 4000 sampler: seed: 42 @@ -31,6 +31,7 @@ sampler: difficulty_weights: {'1': 0.30, '2': 0.25, '3': 0.20, '4': 0.15, '5': 0.10} ambiguity_weights: {minimal: 0.30, low: 0.25, moderate: 0.20, high: 0.15, extreme: 0.10} distractor_weights: {'0': 0.35, '1': 0.30, '2': 0.20, '3': 0.15} + general_text_skill_weight: 0.3 critic: enabled: true provider: @@ -40,7 +41,7 @@ critic: base_url: https://albert.api.etalab.gouv.fr/v1 model: deepseek-v4-flash-0731 model: deepseek-v4-flash-0731 - prompt_version: critic_v1 + prompt_version: critic temperature: 0.0 requests_per_minute: 40 annotator: diff --git a/src/tasksource/jev/synthetic/configs/mock_pilot.yaml b/src/tasksource/jev/synthetic/configs/mock_pilot.yaml index 729fe93..bced59d 100644 --- a/src/tasksource/jev/synthetic/configs/mock_pilot.yaml +++ b/src/tasksource/jev/synthetic/configs/mock_pilot.yaml @@ -13,7 +13,7 @@ generation: concurrency: 8 max_output_tokens: 4000 seed: 42 - prompt_version: generate_v2 + prompt_version: generate n_states: 20 sampler: @@ -32,7 +32,7 @@ critic: base_url: mock:// model: mock-generator model: mock-generator - prompt_version: critic_v1 + prompt_version: critic temperature: 0.0 annotator: diff --git a/src/tasksource/jev/synthetic/configs/openai_luna.yaml b/src/tasksource/jev/synthetic/configs/openai_luna.yaml index 71092d4..8f33b61 100644 --- a/src/tasksource/jev/synthetic/configs/openai_luna.yaml +++ b/src/tasksource/jev/synthetic/configs/openai_luna.yaml @@ -13,7 +13,7 @@ generation: concurrency: 20 max_output_tokens: 4000 seed: 42 - prompt_version: generate_v1 + prompt_version: generate n_states: 1000 sampler: @@ -32,7 +32,7 @@ critic: base_url: https://api.openai.com/v1 model: gpt-6-luna model: gpt-6-luna - prompt_version: critic_v1 + prompt_version: critic temperature: 0.0 annotator: diff --git a/src/tasksource/jev/synthetic/critic.py b/src/tasksource/jev/synthetic/critic.py index 37dc499..e541109 100644 --- a/src/tasksource/jev/synthetic/critic.py +++ b/src/tasksource/jev/synthetic/critic.py @@ -15,6 +15,7 @@ import asyncio import hashlib import json +import re import time from pathlib import Path @@ -46,6 +47,31 @@ def mock_critique(bundle: dict) -> dict: return {"pass": not issues, "issues": issues, "score": 1.0 if not issues else 0.0} +def checked_verdict(response: dict, bundle: dict) -> dict: + """Require one grounded skill/answer check for every live critique.""" + issues = [str(issue) for issue in response.get("issues", [])] + checks = response.get("checks") + questions = bundle.get("questions", []) + expected = {q["question_id"] for q in questions} + if not isinstance(checks, list) or len(checks) != len(questions) or { + check.get("question_id") for check in checks if isinstance(check, dict) + } != expected: + issues.append("missing or duplicate per-question critic checks") + else: + state = re.sub(r"\s+", " ", bundle.get("state", "")).casefold() + for check in checks: + qid = check["question_id"] + quote = check.get("evidence_quote") + if not isinstance(quote, str) or not quote.strip() or re.sub( + r"\s+", " ", quote).strip().casefold() not in state: + issues.append(f"{qid}: evidence quote not found in state") + for field in ("skill_match", "supported", "unique_answer"): + if check.get(field) is not True: + issues.append(f"{qid}: critic {field} check failed") + return {"pass": response.get("pass") is True and not issues, + "issues": issues, "score": float(response.get("score", 0.0))} + + class RequestPacer: """Space critic calls so a batch stays below the provider's minute limit.""" @@ -86,13 +112,10 @@ async def _critique_one(sem, client, model: str, temperature: float, await pacer.wait() result = await providers.chat_complete( client, model, [{"role": "user", "content": prompt}], - temperature=temperature, max_tokens=1000) + temperature=temperature, max_tokens=2000) try: - verdict = extract_json_object(result["text"]) - verdict = {"pass": bool(verdict.get("pass", False)), - "issues": list(verdict.get("issues", [])), - "score": float(verdict.get("score", 0.0))} - except (ValueError, json.JSONDecodeError, TypeError): + verdict = checked_verdict(extract_json_object(result["text"]), bundle) + except (ValueError, json.JSONDecodeError, TypeError, KeyError): verdict = {"pass": False, "issues": ["unparseable critic response"], "score": 0.0} cached.write_text(json.dumps( {"cache_key": key, "state_id": bundle["state_id"], diff --git a/src/tasksource/jev/synthetic/prompts/critic.txt b/src/tasksource/jev/synthetic/prompts/critic.txt new file mode 100644 index 0000000..91ecbbf --- /dev/null +++ b/src/tasksource/jev/synthetic/prompts/critic.txt @@ -0,0 +1,51 @@ +You are a strict quality critic for Jev state bundles (one state, one or more questions). +Valid formats are `choice` (one best option), `noul` (one yes/no proposition), +and `score` (an ordered scale). A bundle with one question is valid. + +BUNDLE: +{{BUNDLE_JSON}} + +Check: +1. All questions genuinely refer to the SAME state (no orphan/hallucinated references). +2. Questions are not paraphrases of each other; each tests a distinct aspect/skill. +3. Questions are not mutually inconsistent and do not leak the answer. +4. Requested distinct formats/skills are actually represented. + In particular, `claim_support` must assess evidence for a specific claim; + `audience_inference` must identify an intended audience, not ticket routing; + `implicit_concern` must be grounded in the concerned person's own words; + and `social_implication` must concern human response, not policy compliance. +5. Each yes/no question asks about a single proposition, and each score question + matches the meaning of its ordered rubric. Reject mismatches such as urgency + with likelihood levels or a numeric scale without defined endpoints. +6. The state belongs to its stated domain and does not mention Jev or the + dataset construction process. +7. Reject a question when the state omits the policy, deadline, current time, + or decision standard needed to answer it. Reject specialist safety or legal + decisions that require outside rules not supplied in the state. +8. For every choice question, derive the answer independently from the state, + then check every option against it. Reject if none matches, if two or more + match, or if the best option depends on an unstated interpretation. Do the + arithmetic for numeric options rather than relying on the wording. +9. Check temporal and numeric questions against the actual dates and numbers. + Reject an event-ordering question that treats a planned deadline as an event + that occurred, or asks for the order of events whose times are not stated. + Reject a comparison that assumes an unreported start, end, or total. +10. Reject reference questions with multiple plausible antecedents and relation + questions that infer a role or relationship absent from the state, even if + it seems plausible from tone or context. + +For EACH question, record a short EXACT quote copied from the state that +supports its answer. Check that the question tests its declared skill and that +one answer is uniquely supported. A quoted sentence about today's conditions +does not establish tomorrow's conditions. A quote that merely makes one option +plausible is insufficient when another option is also plausible. +The skill list is extensible: judge whether each question matches its declared +skill, not whether the skill name appears in this prompt. Keep every issue +short and specific; do not put a reasoning transcript inside an issue string. + +Output STRICT JSON only: +{"pass": bool, "issues": [str], "score": float, + "checks": [{"question_id": str, "evidence_quote": str, + "skill_match": bool, "supported": bool, "unique_answer": bool}]} +Include exactly one check for every question. Set pass to false if any check +fails. The evidence quote must be a contiguous substring of the state. diff --git a/src/tasksource/jev/synthetic/prompts/critic_v1.txt b/src/tasksource/jev/synthetic/prompts/critic_v1.txt deleted file mode 100644 index 753ed45..0000000 --- a/src/tasksource/jev/synthetic/prompts/critic_v1.txt +++ /dev/null @@ -1,22 +0,0 @@ -You are a strict quality critic for Jev state bundles (one state, several questions). - -BUNDLE: -{{BUNDLE_JSON}} - -Check: -1. All questions genuinely refer to the SAME state (no orphan/hallucinated references). -2. Questions are not paraphrases of each other; each tests a distinct aspect/skill. -3. Questions are not mutually inconsistent and do not leak the answer. -4. Requested distinct formats/skills are actually represented. -5. Each yes/no question asks about a single proposition, and each score question - matches the meaning of its ordered rubric. Reject mismatches such as urgency - with likelihood levels or a numeric scale without defined endpoints. -6. The state belongs to its stated domain and does not mention Jev or the - dataset construction process. -7. Reject a question when the state omits the policy, deadline, current time, - or decision standard needed to answer it. Reject specialist safety or legal - decisions that require outside rules not supplied in the state. -8. For each choice question, reject if two or more options are defensible from - the state or if the best option depends on an unstated interpretation. - -Output STRICT JSON only: {"pass": bool, "issues": [str], "score": float}. diff --git a/src/tasksource/jev/synthetic/prompts/generate_v2.txt b/src/tasksource/jev/synthetic/prompts/generate.txt similarity index 54% rename from src/tasksource/jev/synthetic/prompts/generate_v2.txt rename to src/tasksource/jev/synthetic/prompts/generate.txt index 1f83b81..7f708cd 100644 --- a/src/tasksource/jev/synthetic/prompts/generate_v2.txt +++ b/src/tasksource/jev/synthetic/prompts/generate.txt @@ -15,10 +15,35 @@ Rules: authority, include that information explicitly in the state. Do not require outside legal, medical, aviation, or other specialist rules to pick an answer. - Questions must test DISTINCT aspects of the state (not paraphrases). +- General text-analysis questions (topic, emotion, intent, claim support, + stance, document purpose, main point, implicit concern, intended audience, + argument role, stakeholder perspective, social implication, evidence + strength, or message tone) should arise naturally from the same kind of + state as other questions: a report, message, conversation, ticket, or + excerpt. These are reading and interpretation tasks, not arithmetic, + chronology, entity lookup, or literal span retrieval. Include textual cues + for one best answer, but allow realistic nuance. For argument role, identify + whether a passage functions as a claim, support, caveat, or request in the + context. For evidence strength, compare how well stated evidence supports a + specific claim; do not ask for a numerical confidence grade. Claim support + must ask how evidence bears on a specific claim, not restate a message's main + concern. Audience inference asks whom a message addresses or is designed for, + not who should handle a ticket; include an actual addressee or clear audience + cue rather than a generic incident ticket. For implicit concern, use the concerned + person's own words; do not invent a concern for someone absent from the + state. Social implication asks how wording or behavior may affect people, + not what a policy requires next. Include everyday common sense, social + context, and conversational subtext where they fit. Mix straightforward + reading with cases that require combining multiple cues. + Invent varied, context-appropriate options; do not copy labels or wording + from a named benchmark. - For choice questions output EXACTLY n_options options (use the requested count). - Make exactly ONE choice option best supported by the state. Other options may be plausible, but the state must contain evidence that rules them out. Avoid pairs of options that are both valid next steps under the stated policy. +- Before output, solve each choice question from the state and check every + option. The best interpretation must have clear support; alternatives should + differ in meaning rather than be near-synonyms. - For score questions the spec gives ordered "criteria": reuse them VERBATIM, in order, as the "options" array, with matching "min" (first index) and "max" (last index). Never rename, merge, or reorder criteria. State what the diff --git a/src/tasksource/jev/synthetic/prompts/generate_v1.txt b/src/tasksource/jev/synthetic/prompts/generate_v1.txt deleted file mode 100644 index e9a7464..0000000 --- a/src/tasksource/jev/synthetic/prompts/generate_v1.txt +++ /dev/null @@ -1,22 +0,0 @@ -You are a scenario generator for training a decision model called Jev. - -Jev takes a *state* (evidence text) plus typed *questions* (choice / noul / score) -and returns a probability distribution per question. Your job is ONLY to -instantiate the specification below as a plausible scenario. Do NOT output -answers, probabilities, or which option is correct. - -SPEC: -{{SPEC_JSON}} - -Rules: -- Generate ONE coherent state plus EXACTLY the requested number of questions, in order. -- Every question must be answerable or meaningfully assessable from the SAME state. -- Questions must test DISTINCT aspects of the state (not paraphrases). -- For choice questions output EXACTLY n_options options (use the requested count). -- For score questions include "min" and "max" matching the spec. -- For noul questions output only the question text. -- Include requested irrelevant/distractor information naturally in the state. -- Do NOT state or hint which option is correct; do not make one option lexically obvious. -- Do NOT mention difficulty, ambiguity, skill names, evidence, distractors, or any metadata. -- Do NOT include phrases like "correct answer" or "the answer is". -- Output STRICT JSON only, no markdown fences, with keys: {"state": str, "questions": [{"question": str, "options"?: [str], "min"?: int, "max"?: int}]}. diff --git a/src/tasksource/jev/synthetic/prompts/teacher_audit_v1.txt b/src/tasksource/jev/synthetic/prompts/teacher_audit.txt similarity index 100% rename from src/tasksource/jev/synthetic/prompts/teacher_audit_v1.txt rename to src/tasksource/jev/synthetic/prompts/teacher_audit.txt diff --git a/src/tasksource/jev/synthetic/samples/README.md b/src/tasksource/jev/synthetic/samples/README.md index 16be99e..2d97d91 100644 --- a/src/tasksource/jev/synthetic/samples/README.md +++ b/src/tasksource/jev/synthetic/samples/README.md @@ -2,7 +2,7 @@ Live sample from the `pilot_v2_live12` run: **3 state bundles (10 decisions)** generated by `deepseek-v4-flash-0731` via Albert -(OpenAI-compatible), prompt `generate_v2`. Score questions carry their +(OpenAI-compatible), prompt `generate`. Score questions carry their ordered criteria verbatim as `options`, per the Tasksource Jev schema. - `bundles_sample.jsonl` — canonical unit: one row per state diff --git a/src/tasksource/jev/synthetic/specs.py b/src/tasksource/jev/synthetic/specs.py index d73af9f..2f88edb 100644 --- a/src/tasksource/jev/synthetic/specs.py +++ b/src/tasksource/jev/synthetic/specs.py @@ -46,8 +46,25 @@ "owner_assignment", "refund_approval", "access_justification", "data_sensitivity", "compliance_risk", "customer_effort", "resolution_confidence", "contradiction", + "topic_classification", "emotion_recognition", "communicative_intent", + "claim_support", "stance_detection", "document_purpose", "main_point", + "implicit_concern", "audience_inference", "argument_role", + "stakeholder_perspective", "social_implication", "evidence_strength", + "message_tone", ] +# These enter the same domain sampler as every other skill. They are not a +# separate task family or tied to a fixed benchmark label set. +GENERAL_TEXT_SKILLS_BY_FORMAT = { + "choice": ["topic_classification", "emotion_recognition", + "communicative_intent", "claim_support", "stance_detection", + "document_purpose", "main_point", "implicit_concern", + "audience_inference", "argument_role", "stakeholder_perspective", + "social_implication", "evidence_strength", "message_tone"], + "noul": ["claim_support"], + "score": [], +} + # Skill groups keep the latent task space broad; each domain draws from # two or three groups (compatibility without a Cartesian product). SKILL_GROUPS = { @@ -224,9 +241,12 @@ def draw() -> str: def sample_question_spec(rng: random.Random, fmt: str, skill: str) -> dict: if fmt == "choice": - n_options = rng.choices( - [2, 3, 4, 5, 6, 7, 8], - weights=[0.08, 0.22, 0.28, 0.22, 0.12, 0.05, 0.03], k=1)[0] + if skill in GENERAL_TEXT_SKILLS_BY_FORMAT["choice"]: + n_options = rng.choice([3, 4, 5, 6]) + else: + n_options = rng.choices( + [2, 3, 4, 5, 6, 7, 8], + weights=[0.08, 0.22, 0.28, 0.22, 0.12, 0.05, 0.03], k=1)[0] return {"format": "choice", "skill": skill, "n_options": n_options} if fmt == "noul": return {"format": "noul", "skill": skill, @@ -244,6 +264,22 @@ def sample_question_spec(rng: random.Random, fmt: str, skill: str) -> dict: "criteria": rubric} +def sample_skills(rng: random.Random, pool: list[str], formats: list[str], + general_weight: float) -> list[str]: + if not general_weight: + return (rng.sample(pool, len(formats)) if len(formats) <= len(pool) + else [rng.choice(pool) for _ in formats]) + picked = [] + for fmt in formats: + base = [skill for skill in pool if skill not in picked] + general = [skill for skill in GENERAL_TEXT_SKILLS_BY_FORMAT[fmt] + if skill not in picked] + choices = base + general + weights = [1.0] * len(base) + [general_weight] * len(general) + picked.append(rng.choices(choices, weights=weights, k=1)[0]) + return picked + + def sample_spec(index: int, rng: random.Random, sampler_cfg) -> dict: domain = rng.choice(DOMAINS) n = sample_n_questions(rng, sampler_cfg.questions_per_state) @@ -252,8 +288,8 @@ def sample_spec(index: int, rng: random.Random, sampler_cfg) -> dict: sampler_cfg.probability_all_formats_if_n_ge_3) # Distinct skills per state so questions test distinct aspects. pool = domain_skills(domain) - skills = (rng.sample(pool, n) if n <= len(pool) - else [rng.choice(pool) for _ in range(n)]) + skills = sample_skills(rng, pool, formats, + sampler_cfg.general_text_skill_weight) return { "state_id": f"state_{index:06d}", "domain": domain, diff --git a/src/tasksource/jev/synthetic/validate.py b/src/tasksource/jev/synthetic/validate.py index aafb4ed..57f700f 100644 --- a/src/tasksource/jev/synthetic/validate.py +++ b/src/tasksource/jev/synthetic/validate.py @@ -6,8 +6,9 @@ from .schemas import validate_bundle_shape -LEAK_TOKENS = ("difficulty", "ambiguity", "skill", "distractor", - "state_length", "evidence", "certainty") +METADATA_LEAK = re.compile( + r"\b(?:difficulty|ambiguity|skill|distractors?|state_length|" + r"evidence_structure|certainty_hint)\s*[:=]", re.IGNORECASE) ANSWER_LEAK = re.compile(r"correct (answer|option|choice)|answer is\b", re.IGNORECASE) MODEL_LEAK = re.compile(r"\bjev\b", re.IGNORECASE) EITHER_OR_LABEL = re.compile(r"\b(?:positive\s+or\s+negative|negative\s+or\s+positive|yes\s+or\s+no|no\s+or\s+yes)\b", re.IGNORECASE) @@ -37,11 +38,8 @@ def validate_bundle(bundle: dict, spec: dict | None = None) -> list[str]: errors.append("state too short") if len(state) > 8000: errors.append("state too long") - lowered = state.lower() - for token in LEAK_TOKENS: - if token in lowered: - errors.append(f"state leaks sampler metadata: {token}") - break + if METADATA_LEAK.search(state): + errors.append("state leaks sampler metadata field") if ANSWER_LEAK.search(state): errors.append("state leaks correct answer phrasing") if MODEL_LEAK.search(state): diff --git a/tests/test_jev_synthetic.py b/tests/test_jev_synthetic.py index e9515a0..41e7534 100644 --- a/tests/test_jev_synthetic.py +++ b/tests/test_jev_synthetic.py @@ -109,6 +109,16 @@ def test_audited_run_is_broad_with_smaller_score_share(self): self.assertEqual({spec["domain"] for spec in specs}, set(specs_mod.DOMAINS)) self.assertEqual({spec["difficulty"] for spec in specs}, set(specs_mod.DIFFICULTIES)) self.assertEqual({spec["ambiguity"] for spec in specs}, set(specs_mod.AMBIGUITY_LEVELS)) + general = set().union(*specs_mod.GENERAL_TEXT_SKILLS_BY_FORMAT.values()) + general_rows = [(spec["domain"], q["skill"], q["format"]) + for spec in specs for q in spec["questions"] + if q["skill"] in general] + self.assertTrue(0.08 < len(general_rows) / sum(formats.values()) < 0.16) + self.assertEqual({skill for _, skill, _ in general_rows}, general) + self.assertTrue(all(fmt in {"choice", "noul"} for _, _, fmt in general_rows)) + for skill in general: + self.assertGreaterEqual(len({domain for domain, sampled, _ in general_rows + if sampled == skill}), 15) class BundleTest(unittest.TestCase): @@ -155,6 +165,18 @@ def test_rejects_pilot_format_and_model_leaks(self): self.assertTrue(any("likelihood levels for urgency" in error for error in errors)) self.assertTrue(any("target model" in error for error in errors)) + def test_natural_metadata_words_are_not_rejected(self): + bundle = { + "state_id": "s", + "state": "The analyst found evidence of difficulty completing the claim review.", + "questions": [{"question_id": "q0", "format": "noul", + "question": "Is the claim review complete?"}], + } + self.assertEqual(validate_mod.validate_bundle(bundle), []) + bundle["state"] += " Difficulty: 4." + self.assertIn("state leaks sampler metadata field", + validate_mod.validate_bundle(bundle)) + def test_flat_preserves_grouping(self): spec = specs_mod.sample_specs(_cfg().sampler, 5)[0] bundle = mock_realization(spec) @@ -225,6 +247,22 @@ def test_unknown_annotator_rejected(self): class CriticTest(unittest.TestCase): + def test_live_critic_requires_grounded_check_for_each_question(self): + bundle = {"state": "The customer says the package arrived damaged yesterday.", + "questions": [{"question_id": "q0"}, {"question_id": "q1"}]} + check = lambda qid, quote: {"question_id": qid, + "evidence_quote": quote, + "skill_match": True, "supported": True, + "unique_answer": True} + response = {"pass": True, "issues": [], "score": 1.0, + "checks": [check("q0", "package arrived damaged"), + check("q1", "The customer says")]} + self.assertTrue(critic_mod.checked_verdict(response, bundle)["pass"]) + response["checks"][1]["evidence_quote"] = "invented evidence" + self.assertFalse(critic_mod.checked_verdict(response, bundle)["pass"]) + response["checks"] = response["checks"][:1] + self.assertFalse(critic_mod.checked_verdict(response, bundle)["pass"]) + def test_critic_uses_own_provider(self): cfg = _cfg() from tasksource.jev.synthetic.config import ProviderConfig