Choose how to run this agent
Download Agent
Choose how you want to use this agent:
Use Security Settings Key (Recommended)
Use the API key you've already saved in Security Settings. Quick and convenient!
- No need to re-enter API key
- Works offline after download
- Centralized key management
No API key found in Security Settings. Add one now
Enter API Key Manually
Enter your API key now for this specific agent download.
- Use different key for this agent
- One-time use (not saved)
- Works offline after download
Configure Agent Encryption
Description
Based on the Annual Benefit Limit Extractor template.
What this agent can do
My Annual Benefit Limit Extractor is built from the Annual Benefit Limit Extractor template and loads pypdf in the browser. Runs fully on your own device: llama.cpp compiled to WebAssembly, GPU-accelerated through WebGPU, with no API key and no server. After the one-time model download it works offline. Can run on OpenAI models with your own API key, encrypted in your browser. Can run on Anthropic Claude models with your own API key, encrypted in your browser.
Source Code
import io
import re
from pypdf import PdfReader
_documents = []
def _reset_documents():
_documents.clear()
def _add_pdf(name, raw):
try:
reader = PdfReader(io.BytesIO(bytes(raw)))
if reader.is_encrypted:
raise ValueError('This PDF is password-protected.')
pages = []
for number, page in enumerate(reader.pages, 1):
content = page.extract_text(extraction_mode='layout') or ''
content = re.sub(r'[ \t]+', ' ', content)
pages.append((number, content))
if sum(len(text.strip()) for _, text in pages) < 40:
raise ValueError('No selectable text found. Scanned PDFs need OCR before upload.')
_documents.append((str(name), pages))
return str(len(pages)) + ' pages loaded'
except ValueError as exc:
raise ValueError(str(name) + ': ' + str(exc)) from exc
except Exception as exc:
raise ValueError(str(name) + ': Unable to read the PDF; confirm it is a valid text-based brochure.') from exc
def _evidence(pages):
terms = re.compile(r'annual benefit|annual limit|annual maximum|overall limit|yearly limit|per (?:policy )?year|per annum|sum insured|maximum benefit|benefit limit', re.I)
candidates = []
for number, text in pages:
for match in list(terms.finditer(text))[:28]:
start = max(0, match.start() - 260)
end = min(len(text), match.end() + 650)
fragment = text[start:end].strip()
weight = 2 if re.search(r'annual|overall|yearly|per annum', match.group(), re.I) else 1
candidates.append((weight, number, match.start(), fragment))
candidates.sort(key=lambda item: (-item[0], item[1], item[2]))
excerpts = []
used = set()
size = 0
for _, number, position, fragment in candidates:
key = (number, position // 420)
if key in used:
continue
used.add(key)
part = '[Page ' + str(number) + '] ' + fragment[:1050]
if size + len(part) > 5400:
break
excerpts.append(part)
size += len(part)
if not excerpts:
excerpts = ['[Page ' + str(number) + '] ' + text[:1350] for number, text in pages if text.strip()][:3]
return '\n\n'.join(excerpts)[:5500]
def _format_answer(name, answer):
rows = []
for line in str(answer).splitlines():
line = line.strip().strip('`* ')
if not line or 'LIMIT' not in line.upper():
continue
match = re.search(r'LIMIT\s*[:=]\s*(NOT_FOUND|[0-9][0-9, ]*(?:\.00)?)\b', line, re.I)
if not match:
continue
raw = match.group(1)
normalized = 'NOT_FOUND' if raw.upper() == 'NOT_FOUND' else raw.replace(',', '').replace(' ', '').removesuffix('.00')
if normalized != 'NOT_FOUND' and not normalized.isdigit():
continue
rows.append(re.sub(r'LIMIT\s*[:=]\s*(NOT_FOUND|[0-9][0-9, ]*(?:\.00)?)\b', 'LIMIT=' + normalized, line, count=1, flags=re.I)[:900])
if not rows:
rows = ['LIMIT=NOT_FOUND | No unambiguous annual benefit limit confirmed in extracted PDF text.']
return 'BROCHURE: ' + name + '\n' + '\n'.join(rows[:12])
async def process_user_query(query: str) -> str:
if len(_documents) < 2:
return 'Upload at least two product brochure PDFs first.'
instruction = str(query).strip()[:350]
reports = []
for name, pages in _documents:
prompt = ('Read ONLY the quoted PDF excerpts below. Find the overall annual benefit limit, not a specific treatment sublimit, deductible, or premium. '
'If the brochure explicitly lists multiple plan tiers, provide one row per tier. '
'Each row MUST follow this exact format: PLAN=<name or overall> | LIMIT=<integer digits only or NOT_FOUND> | CURRENCY=<currency or unknown> | PAGE=<page number or unknown> | EVIDENCE=<short literal excerpt>. '
'Do not invent amounts. If the overall annual limit is absent or ambiguous, return LIMIT=NOT_FOUND. '
'Do not mistake annual wording for a monetary amount. Use only evidence shown. '
'Additional extraction focus: ' + instruction + '\nPDF: ' + name + '\nEXCERPTS:\n' + _evidence(pages))
answer = await agentop_llm.generate(prompt, globals().get('TEMPLATE_SYSTEM_PROMPT', ''))
reports.append(_format_answer(name, answer))
return '\n=== BROCHURE ===\n'.join(reports)