jaai-ebr-001-en
Evidence-based research
Separate a fictional statement of intent from evidence of an outcome.
Prompt
Assess only the supplied fictional claims and source spans. Distinguish a planned target from a measured result and missing documentation from failure. Return the required JSON with claim IDs and evidence IDs; do not browse or infer a national ranking.
Synthetic inputs
{
"synthetic": true,
"sources": [
{
"source_id": "S1",
"uri": "urn:steadywrk:synthetic:jaai:0.1:S1",
"provenance": "Original fictional source; not a report about a real organization.",
"text_en": "Fictional Workshop Group W-17 memo, 1 September 2026: A trial is planned at two fictional workshops from 15 to 30 September. The target is to reduce invoice review time from 40 to 30 minutes. This memo contains no measured result.",
"text_ar": "مذكرة مجموعة الورش الافتراضية W-17، 1 أيلول 2026: من المخطط تجربة النظام في ورشتين افتراضيتين من 15 إلى 30 أيلول. الهدف خفض وقت مراجعة الفاتورة من 40 إلى 30 دقيقة. لا تتضمن المذكرة نتيجة مقاسة.",
"text_sha256": {
"en": "ac2e22018cfa084ab199001b20641a144a6481d7591df0af7f91827f31664faa",
"ar": "0209cce07bb02be1d05408a827b7a3e8a57d82b802678357f708604e44b0b274"
}
},
{
"source_id": "S2",
"uri": "urn:steadywrk:synthetic:jaai:0.1:S2",
"provenance": "Original fictional source; not a report about a real organization.",
"text_en": "Fictional trial log, 15–21 September 2026: Eighteen draft invoices were recorded, each marked as reviewed by a person. No baseline measurements or review durations were recorded.",
"text_ar": "سجل التجربة الافتراضي، 15–21 أيلول 2026: سُجّلت 18 مسودة فاتورة، وعُلّمت كل مسودة بأنها خضعت لمراجعة بشرية. لم تُسجّل قياسات لخط الأساس أو مدة المراجعة.",
"text_sha256": {
"en": "dff8c697bc04c954274fa37c46cf50d7b00ad427ddaed0ca079db5fae0489a4c",
"ar": "34558ea97567667d0ceac5cb1ab1f5cc12d43dc3c1f0706748c61a12707f9429"
}
},
{
"source_id": "S3",
"uri": "urn:steadywrk:synthetic:jaai:0.1:S3",
"provenance": "Original fictional source; not a report about a real organization.",
"text_en": "Fictional evidence register, 1 October 2026: No final completion record for 30 September and no independent assessment have been supplied. Absence from this register does not establish that the trial failed.",
"text_ar": "سجل الأدلة الافتراضي، 1 تشرين الأول 2026: لم يُقدّم سجل نهائي يؤكد إتمام التجربة في 30 أيلول، ولم يُقدّم تقييم مستقل. عدم وجود هذه الوثائق في السجل لا يثبت فشل التجربة.",
"text_sha256": {
"en": "b4887dc59f471efda9054125c4ab77748a8f9390f7ce277fb3f9d96b6ef4ef43",
"ar": "8dc6d46c13e59727b478a84687d057f48dff5734cbe1f0124b6cb4f808ab1d13"
}
}
],
"candidate_claims": [
{
"claim_id": "C1",
"text_en": "A trial was planned for two workshops.",
"text_ar": "كان من المخطط إجراء تجربة في ورشتين."
},
{
"claim_id": "C2",
"text_en": "The supplied log records 18 drafts with human review markers.",
"text_ar": "يسجل السجل المقدم 18 مسودة عليها علامات مراجعة بشرية."
},
{
"claim_id": "C3",
"text_en": "Measured review time fell from 40 to 30 minutes.",
"text_ar": "انخفض وقت المراجعة المقاس من 40 إلى 30 دقيقة."
},
{
"claim_id": "C4",
"text_en": "The system is the most accurate AI system in Jordan.",
"text_ar": "النظام هو أدق نظام ذكاء اصطناعي في الأردن."
},
{
"claim_id": "C5",
"text_en": "The trial was confirmed complete on 30 September.",
"text_ar": "تأكد إتمام التجربة في 30 أيلول."
}
],
"rule": "A planned target is not a measured outcome. Treat a claim as supported only when a supplied source directly establishes it."
}Required output schema
{
"format": "json",
"schema": {
"type": "object",
"properties": {
"supported_claim_ids": {
"type": "array",
"items": {
"type": "string",
"enum": [
"C1",
"C2",
"C3",
"C4",
"C5"
]
},
"uniqueItems": true
},
"unsupported_claim_ids": {
"type": "array",
"items": {
"type": "string",
"enum": [
"C1",
"C2",
"C3",
"C4",
"C5"
]
},
"uniqueItems": true
},
"evidence": {
"type": "object",
"properties": {
"C1": {
"type": "array",
"items": {
"type": "string",
"enum": [
"S1",
"S2",
"S3"
]
},
"uniqueItems": true
},
"C2": {
"type": "array",
"items": {
"type": "string",
"enum": [
"S1",
"S2",
"S3"
]
},
"uniqueItems": true
}
},
"required": [
"C1",
"C2"
],
"additionalProperties": false
},
"outcome_status": {
"type": "string",
"enum": [
"established",
"insufficient_evidence"
]
},
"unproven_outcome_claim_ids": {
"type": "array",
"items": {
"type": "string",
"enum": [
"C1",
"C2",
"C3",
"C4",
"C5"
]
},
"uniqueItems": true
}
},
"required": [
"supported_claim_ids",
"unsupported_claim_ids",
"evidence",
"outcome_status",
"unproven_outcome_claim_ids"
],
"additionalProperties": false
},
"additional_prose_allowed": false
}Reference assertions
[
{
"assertion_id": "EBR-1",
"check": "set_equals",
"path": "/supported_claim_ids",
"expected": [
"C1",
"C2"
],
"weight": 1
},
{
"assertion_id": "EBR-2",
"check": "set_equals",
"path": "/unsupported_claim_ids",
"expected": [
"C3",
"C4",
"C5"
],
"weight": 1
},
{
"assertion_id": "EBR-3",
"check": "set_equals",
"path": "/evidence/C1",
"expected": [
"S1"
],
"weight": 1
},
{
"assertion_id": "EBR-4",
"check": "set_equals",
"path": "/evidence/C2",
"expected": [
"S2"
],
"weight": 1
},
{
"assertion_id": "EBR-5",
"check": "equals",
"path": "/outcome_status",
"expected": "insufficient_evidence",
"weight": 1
},
{
"assertion_id": "EBR-6",
"check": "set_equals",
"path": "/unproven_outcome_claim_ids",
"expected": [
"C3",
"C4",
"C5"
],
"weight": 1
}
]Scoring rule
{
"version": "0.1",
"aggregation": "passed_assertion_weights / total_assertion_weights",
"max_score": 6,
"invalid_output": "All assertions fail if the response is not one JSON object satisfying required_output.schema.",
"review_required": false,
"report": [
"passed_assertions",
"total_assertions",
"weighted_score",
"output_schema_valid"
],
"reference_output": {
"supported_claim_ids": [
"C1",
"C2"
],
"unsupported_claim_ids": [
"C3",
"C4",
"C5"
],
"evidence": {
"C1": [
"S1"
],
"C2": [
"S2"
]
},
"outcome_status": "insufficient_evidence",
"unproven_outcome_claim_ids": [
"C3",
"C4",
"C5"
]
}
}When to abstain
[
{
"condition": "No measured durations or comparison data are supplied.",
"required_action": "Mark C3 and C4 unsupported; do not claim demonstrated improvement or a national ranking."
},
{
"condition": "The completion record is missing.",
"required_action": "Mark C5 unsupported; do not infer trial failure or completion."
}
]Practice connection: Applied AI Research & Evaluation
Explore this area