"""Offline invoice extraction workflow. No network, API key, or billed requests. The adapter seam returns AdapterReply; replace FixtureAdapter with a provider adapter only after testing delivery, stop-reason and usage handling separately. Acceptance is deliberately limited to canonical labeled text, not arbitrary OCR. """ import argparse from dataclasses import dataclass from datetime import date from decimal import Decimal import hashlib import json import re FIELDS = { 'invoice_id': 'Invoice ID', 'vendor': 'Vendor', 'currency': 'Currency', 'total_due': 'Total due', 'due_date': 'Due date', } DOCUMENT = '\n'.join([ 'Invoice ID: INV-1042', 'Vendor: Cedar Research', 'Currency: USD', 'Subtotal: 240.00', 'Tax: 12.00', 'Total due: 252.00', 'Due date: 2026-11-08', ]) @dataclass(frozen=True) class AdapterReply: text: str stop_reason: str = 'end_turn' @dataclass(frozen=True) class Issue: code: str field: str message: str retryable: bool def _unique_object(pairs): result = {} for key, value in pairs: if key in result: raise ValueError('Duplicate JSON key') result[key] = value return result def _reject_constant(value): raise ValueError('Nonstandard JSON constant') def canonical_values(document, label): # Only this narrow, case-sensitive label format is accepted automatically. matches = [] for number, line in enumerate(document.splitlines(), 1): match = re.fullmatch(re.escape(label) + r': (.+)', line) if match: matches.append((number, line, match.group(1))) return matches def validate(document, raw): """Return (record, issues); record is None whenever any issue remains.""" issues = [] if not isinstance(raw, str) or len(raw) > 50000: return None, [Issue('output_limit', '', 'Response is not bounded text.', False)] try: candidate = json.loads(raw, object_pairs_hook=_unique_object, parse_constant=_reject_constant) except RecursionError: return None, [Issue('parser_depth', '', 'JSON exceeds parser nesting limits.', False)] except (ValueError, TypeError): return None, [Issue('invalid_json', '', 'Return one strict JSON object.', True)] if not isinstance(candidate, dict) or set(candidate) != {'fields'}: return None, [Issue('schema', '', 'Only the top-level fields object is allowed.', True)] fields = candidate['fields'] if not isinstance(fields, dict) or set(fields) != set(FIELDS): return None, [Issue('schema', '', 'Return exactly the five required fields.', True)] lines = document.splitlines() clean = {} evidence = {} for name, label in FIELDS.items(): item = fields[name] if not isinstance(item, dict) or set(item) != {'value', 'evidence'}: issues.append(Issue('schema', name, 'Use value and evidence only.', True)) continue value, support = item['value'], item['evidence'] if not isinstance(value, str) or not value or value != value.strip(): issues.append(Issue('schema', name, 'Use a nonempty unpadded string.', True)) continue if not isinstance(support, list) or len(support) != 1: issues.append(Issue('schema', name, 'Use one evidence object.', True)) continue citation = support[0] if not isinstance(citation, dict) or set(citation) != {'line', 'quote'}: issues.append(Issue('schema', name, 'Evidence requires line and quote.', True)) continue number, quote = citation['line'], citation['quote'] if type(number) is not int or not isinstance(quote, str): issues.append(Issue('schema', name, 'Line must be an integer and quote text.', True)) continue if not 1 <= number <= len(lines) or lines[number - 1] != quote: issues.append(Issue('evidence', name, 'Citation does not match the source.', False)) continue matches = canonical_values(document, label) if len(matches) != 1: issues.append(Issue('ambiguous_source', name, 'Expected one canonical source label.', False)) continue if matches[0] != (number, quote, value): issues.append(Issue('unsupported_value', name, 'Value is not the value under its source label.', False)) continue valid = True if name == 'invoice_id': valid = bool(re.fullmatch(r'[A-Za-z0-9][A-Za-z0-9_-]{0,63}', value)) elif name == 'vendor': valid = len(value) <= 120 and value.isprintable() elif name == 'currency': valid = value in {'USD', 'EUR', 'GBP', 'INR'} elif name == 'total_due': valid = bool(re.fullmatch(r'(?:0|[1-9][0-9]{0,8})\.[0-9]{2}', value)) if valid: valid = Decimal(value) <= Decimal('999999999.99') elif name == 'due_date': try: valid = bool(re.fullmatch(r'[0-9]{4}-[0-9]{2}-[0-9]{2}', value)) if valid: date.fromisoformat(value) except ValueError: valid = False if not valid: issues.append(Issue('business_rule', name, 'Source value is outside this policy.', False)) continue clean[name] = value evidence[name] = dict(citation) if issues: return None, issues return {'values': clean, 'evidence': evidence, 'document_sha256': hashlib.sha256(document.encode('utf-8')).hexdigest(), 'policy_version': 'canonical-invoice-v1'}, [] def run_workflow(document, adapter, max_attempts=2): """One repair is allowed only for a completed reply with structural errors.""" if type(max_attempts) is not int or not 1 <= max_attempts <= 2: raise ValueError('max_attempts must be 1 or 2') feedback = [] history = [] for attempt in range(1, max_attempts + 1): try: reply = adapter.extract(document, feedback) except Exception: # Delivery/charging status may be unknown. Do not blindly repeat. history.append({'attempt': attempt, 'codes': ['adapter_failure']}) return {'state': 'review', 'record': None, 'history': history} if not isinstance(reply, AdapterReply) or reply.stop_reason != 'end_turn': history.append({'attempt': attempt, 'codes': ['incomplete_or_refused']}) return {'state': 'review', 'record': None, 'history': history} record, issues = validate(document, reply.text) history.append({'attempt': attempt, 'codes': [issue.code for issue in issues]}) if record is not None: return {'state': 'accepted', 'record': record, 'history': history} if any(not issue.retryable for issue in issues) or attempt == max_attempts: return {'state': 'review', 'record': None, 'history': history} feedback = [{'field': issue.field, 'code': issue.code, 'message': issue.message} for issue in issues] raise AssertionError('Unreachable') def fixture_candidate(document=DOCUMENT): """A fixture constructor, NOT a model or a general extraction algorithm.""" fields = {} for name, label in FIELDS.items(): number, quote, value = canonical_values(document, label)[0] fields[name] = {'value': value, 'evidence': [{'line': number, 'quote': quote}]} return {'fields': fields} class FixtureAdapter: def __init__(self, replies): self.replies = iter(replies) self.feedback_seen = [] def extract(self, document, feedback): self.feedback_seen.append(feedback) return next(self.replies) def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument('--scenario', choices=['accepted', 'repair', 'review'], default='accepted') args = parser.parse_args() candidate = fixture_candidate() good = AdapterReply(json.dumps(candidate)) if args.scenario == 'repair': replies = [AdapterReply('{not valid JSON'), good] elif args.scenario == 'review': candidate['fields']['total_due']['value'] = '240.00' candidate['fields']['total_due']['evidence'] = [{'line': 4, 'quote': 'Subtotal: 240.00'}] replies = [AdapterReply(json.dumps(candidate))] else: replies = [good] print(json.dumps(run_workflow(DOCUMENT, FixtureAdapter(replies)), indent=2)) if __name__ == '__main__': main()