{
  "publisher": "Praevideo",
  "template_version": "1.0",
  "artifact_type": "task-design-template",
  "usage": "Adapt with domain experts before evaluation. Contains no cases, model scores, or executable verifiers.",
  "url": "https://praevideogroup.com/benchmarks/#professional-services",
  "title": "Explain a reconciliation difference",
  "inputs": [
    "Ledger entries",
    "Statement entries",
    "Supporting documents",
    "Accounting period",
    "Matching rules and tolerances"
  ],
  "output": [
    "Matched entries with source references",
    "Unmatched entries",
    "Arithmetic reconciliation",
    "Explanation of remaining difference",
    "Items requiring review"
  ],
  "hard_failures": [
    "Invented supporting evidence",
    "Unsupported adjustment",
    "Arithmetic does not reconcile",
    "Posting an adjustment without approval"
  ],
  "quality_measures": [
    "Correct matches",
    "Supported explanations",
    "Appropriate escalation of unresolved entries"
  ],
  "edge_cases": [
    "Duplicate references",
    "Partial payments",
    "Timing differences",
    "Missing supporting documents"
  ],
  "action_boundary": "Propose outputs in a test environment. No production writes.",
  "review_before_use": [
    "Confirm task relevance and authorized source material",
    "Define output schema and allowed tools",
    "Implement outcome checks and validate them with experts",
    "Separate development, training, and held-out evaluation cases",
    "Check split boundaries for duplicate customers, assets, incidents, and templates",
    "Record model, prompt, tools, environment, retries, latency, cost, and human review",
    "Distinguish model failures from evaluation infrastructure failures"
  ]
}
