// agent

My Annual Benefit Limit Extractor

by gary · Sep 29, 2026 Public

Choose how to run this agent

⚡ Local runs on your GPU. For usable speed it needs a WebGPU-capable browser — Chrome or Edge on a machine with a graphics card, or an Apple Silicon Mac. Without a supported GPU, pick OpenAI or Anthropic above instead. Check your machine
Try it now

Requires an API key and an AgentOp account.

9 downloads
0 forks
0.0 rating

Description

Based on the Annual Benefit Limit Extractor template.

What this agent can do

My Annual Benefit Limit Extractor is built from the Annual Benefit Limit Extractor template and loads pypdf in the browser. Runs fully on your own device: llama.cpp compiled to WebAssembly, GPU-accelerated through WebGPU, with no API key and no server. After the one-time model download it works offline. Can run on OpenAI models with your own API key, encrypted in your browser. Can run on Anthropic Claude models with your own API key, encrypted in your browser.

Source Code

agent.py
import io
import re
from pypdf import PdfReader

_documents = []


def _reset_documents():
    _documents.clear()


def _add_pdf(name, raw):
    try:
        reader = PdfReader(io.BytesIO(bytes(raw)))
        if reader.is_encrypted:
            raise ValueError('This PDF is password-protected.')
        pages = []
        for number, page in enumerate(reader.pages, 1):
            content = page.extract_text(extraction_mode='layout') or ''
            content = re.sub(r'[ \t]+', ' ', content)
            pages.append((number, content))
        if sum(len(text.strip()) for _, text in pages) < 40:
            raise ValueError('No selectable text found. Scanned PDFs need OCR before upload.')
        _documents.append((str(name), pages))
        return str(len(pages)) + ' pages loaded'
    except ValueError as exc:
        raise ValueError(str(name) + ': ' + str(exc)) from exc
    except Exception as exc:
        raise ValueError(str(name) + ': Unable to read the PDF; confirm it is a valid text-based brochure.') from exc


def _evidence(pages):
    terms = re.compile(r'annual benefit|annual limit|annual maximum|overall limit|yearly limit|per (?:policy )?year|per annum|sum insured|maximum benefit|benefit limit', re.I)
    candidates = []
    for number, text in pages:
        for match in list(terms.finditer(text))[:28]:
            start = max(0, match.start() - 260)
            end = min(len(text), match.end() + 650)
            fragment = text[start:end].strip()
            weight = 2 if re.search(r'annual|overall|yearly|per annum', match.group(), re.I) else 1
            candidates.append((weight, number, match.start(), fragment))
    candidates.sort(key=lambda item: (-item[0], item[1], item[2]))
    excerpts = []
    used = set()
    size = 0
    for _, number, position, fragment in candidates:
        key = (number, position // 420)
        if key in used:
            continue
        used.add(key)
        part = '[Page ' + str(number) + '] ' + fragment[:1050]
        if size + len(part) > 5400:
            break
        excerpts.append(part)
        size += len(part)
    if not excerpts:
        excerpts = ['[Page ' + str(number) + '] ' + text[:1350] for number, text in pages if text.strip()][:3]
    return '\n\n'.join(excerpts)[:5500]


def _format_answer(name, answer):
    rows = []
    for line in str(answer).splitlines():
        line = line.strip().strip('`* ')
        if not line or 'LIMIT' not in line.upper():
            continue
        match = re.search(r'LIMIT\s*[:=]\s*(NOT_FOUND|[0-9][0-9, ]*(?:\.00)?)\b', line, re.I)
        if not match:
            continue
        raw = match.group(1)
        normalized = 'NOT_FOUND' if raw.upper() == 'NOT_FOUND' else raw.replace(',', '').replace(' ', '').removesuffix('.00')
        if normalized != 'NOT_FOUND' and not normalized.isdigit():
            continue
        rows.append(re.sub(r'LIMIT\s*[:=]\s*(NOT_FOUND|[0-9][0-9, ]*(?:\.00)?)\b', 'LIMIT=' + normalized, line, count=1, flags=re.I)[:900])
    if not rows:
        rows = ['LIMIT=NOT_FOUND | No unambiguous annual benefit limit confirmed in extracted PDF text.']
    return 'BROCHURE: ' + name + '\n' + '\n'.join(rows[:12])


async def process_user_query(query: str) -> str:
    if len(_documents) < 2:
        return 'Upload at least two product brochure PDFs first.'
    instruction = str(query).strip()[:350]
    reports = []
    for name, pages in _documents:
        prompt = ('Read ONLY the quoted PDF excerpts below. Find the overall annual benefit limit, not a specific treatment sublimit, deductible, or premium. '
                  'If the brochure explicitly lists multiple plan tiers, provide one row per tier. '
                  'Each row MUST follow this exact format: PLAN=<name or overall> | LIMIT=<integer digits only or NOT_FOUND> | CURRENCY=<currency or unknown> | PAGE=<page number or unknown> | EVIDENCE=<short literal excerpt>. '
                  'Do not invent amounts. If the overall annual limit is absent or ambiguous, return LIMIT=NOT_FOUND. '
                  'Do not mistake annual wording for a monetary amount. Use only evidence shown. '
                  'Additional extraction focus: ' + instruction + '\nPDF: ' + name + '\nEXCERPTS:\n' + _evidence(pages))
        answer = await agentop_llm.generate(prompt, globals().get('TEMPLATE_SYSTEM_PROMPT', ''))
        reports.append(_format_answer(name, answer))
    return '\n=== BROCHURE ===\n'.join(reports)