Skip to content
Open
Show file tree
Hide file tree
Changes from 5 commits
Commits
Show all changes
32 commits
Select commit Hold shift + click to select a range
d798bfb
fix: import Union in main.py and correct pytest directory in Makefile
Mar 31, 2026
6fd3496
feat(pdf): Map Boolean Checkbox and Radio States & Fix main.py NameError
Mar 31, 2026
15122f6
Merge branch 'upstream/main' into fix/startup-and-tests
Dotify71 May 15, 2026
9feeb78
feat(pdf): enforce strict boolean typing for checkbox/radio fields
Dotify71 May 15, 2026
6977124
Merge upstream/main and resolve conflicts
Dotify71 May 24, 2026
a10663d
fix: add strict boolean schema validation and retry loop
Dotify71 May 26, 2026
c525c9b
Merge branch 'main' into fix/startup-and-tests
Dotify71 Jun 13, 2026
b27e5b7
refactor: replace Union with modern syntax and remove unused method call
Dotify71 Jul 4, 2026
0a959b0
fix: remove trailing whitespace in main.py
Dotify71 Jul 4, 2026
19b16c2
chore: fix f-string lint error
Dotify71 Jul 13, 2026
f4a21eb
fix: resolve ruff linter errors
Dotify71 Jul 28, 2026
7ebb951
fix: resolve ruff lint errors (I001 and SIM102)
Dotify71 Jul 30, 2026
fd57583
feat: :sparkles: first implementation of the benchmark, including fro…
marcvergees Aug 1, 2026
91fd496
fix: :bug: fixing test passing with empty structure of everything
marcvergees Aug 3, 2026
4927d9a
style: :lipstick: adding notes for guys
marcvergees Aug 3, 2026
73a382c
feat: add large-model reference evaluator
vharkins1 Aug 12, 2026
684ebfa
ics201 & 202
marcvergees Aug 13, 2026
fe000b1
ics203
marcvergees Aug 13, 2026
d9f0499
ics204
marcvergees Aug 13, 2026
a3ab05a
ics205 & ics205a
marcvergees Aug 13, 2026
2f24ff8
ics 206 & ics 213
marcvergees Aug 13, 2026
559030d
ics207 & ics 208
marcvergees Aug 13, 2026
2a6cd2b
ICS dataset generation markdown
marcvergees Aug 13, 2026
ce908a5
Merge pull request #663 from fireform-core/659-dataset-creations
vharkins1 Aug 13, 2026
2dd9e3a
refactor: :recycle: linting errors
marcvergees Aug 13, 2026
157e01a
linter errors 2
marcvergees Aug 13, 2026
86ade1b
Merge pull request #662 from fireform-core/612-feat-benchmark-module-…
marcvergees Aug 13, 2026
cb25b47
feat: :sparkles: add ics201-208 & 213 pdfs
marcvergees Aug 14, 2026
13806d2
refactor: :recycle: delete reference_benchmarks
marcvergees Aug 14, 2026
2378c26
feat: :sparkles: implementation of pdfs in the runner
marcvergees Aug 14, 2026
57d3d3b
Merge pull request #668 from fireform-core/667-add-pdfs-to-benchmark
marcvergees Aug 14, 2026
d71e765
Merge development and resolve conflicts
Dotify71 Aug 26, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
31 changes: 29 additions & 2 deletions src/filler.py
Original file line number Diff line number Diff line change
Expand Up @@ -39,8 +39,35 @@ def fill_form(self, pdf_form: str, llm: LLM):
for annot in sorted_annots:
if annot.Subtype == "/Widget" and annot.T:
if i < len(answers_list):
annot.V = f"{answers_list[i]}"
annot.AP = None
answer = answers_list[i]

# Check if the field type is a Button (Checkbox/Radio)
field_type = annot.FT if annot.FT else (annot.Parent.FT if annot.Parent else None)
if str(field_type) == "/Btn":
# The LLM pipeline guarantees Python bool for boolean fields.
# We check isinstance(answer, bool) so only an explicit True
# activates the button — no fuzzy string matching needed.
is_truthy = isinstance(answer, bool) and answer

# Find the 'ON' state from the appearance dictionary
on_state = "/Yes" # Default assumption
if annot.AP and annot.AP.N:
keys = [k for k in annot.AP.N.keys() if k != "/Off"]
if keys:
on_state = keys[0]

if is_truthy:
from pdfrw import PdfName
annot.V = PdfName(on_state.strip("/"))
annot.AS = PdfName(on_state.strip("/"))
else:
from pdfrw import PdfName
annot.V = PdfName("Off")
annot.AS = PdfName("Off")
else:
annot.V = f"{answer}"
annot.AP = None

i += 1
else:
# Stop if we run out of answers
Expand Down
57 changes: 47 additions & 10 deletions src/llm.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,15 +10,31 @@ def __init__(self, transcript_text: str=None, target_fields: list=None, json_dic
self._target_fields = target_fields
self._json = json_dict if json_dict is not None else {}

def build_prompt(self, current_field: str, current_type: str = "string"):
def build_prompt(self, current_field: str, field_type: type = str):
"""
This method is in charge of the prompt engineering. It creates a specific prompt for each target field.
@params: current_field -> represents the current element of the json that is being prompted.
@params: current_type -> hint to the LLM about the expected value shape (date, number, etc.).
This method is in charge of the prompt engineering. It creates a specific prompt
for each target field, taking into account the expected field type.

If the field type is `bool`, the LLM is explicitly instructed to return only
the literal string `True` or `False` — no fuzzy values like 'yes' or '1'.

@params:
current_field -> the name of the JSON field to extract.
field_type -> the expected Python type (e.g. str, bool).
"""
prompt_path = os.path.join(os.path.dirname(__file__), "prompt.txt")
with open(prompt_path, "r") as f:
template = f.read()

current_type = "boolean" if field_type is bool else "string"

if field_type is bool:
bool_instruction = (
"\nIMPORTANT: This field is a boolean. "
"You MUST respond with ONLY the literal word True or False. "
"Do not use 'yes', 'no', '1', '0', or any other value."
)
return template.format(field=current_field, type=current_type, text=self._transcript_text) + bool_instruction

return template.format(field=current_field, type=current_type, text=self._transcript_text)

Expand All @@ -27,8 +43,9 @@ def main_loop(self):
max_retries = 3

total_fields = len(self._target_fields)
for i, (field, field_type) in enumerate(self._target_fields.items(), 1):
prompt = self.build_prompt(field, field_type if isinstance(field_type, str) else "string")
for i, (field, field_val) in enumerate(self._target_fields.items(), 1):
field_type = field_val if isinstance(field_val, type) else str
prompt = self.build_prompt(field, field_type=field_type)
ollama_host = os.getenv("OLLAMA_HOST", "http://localhost:11434").rstrip("/")
ollama_url = f"{ollama_host}/api/generate"

Expand Down Expand Up @@ -74,15 +91,35 @@ def main_loop(self):

def add_response_to_json(self, field: str, value: str):
"""
this method adds the following value under the specified field,
or under a new field if the field doesn't exist, to the json dict
Adds the LLM response under the specified field in the JSON dict.

If the field type in _target_fields is `bool`, the response is strictly
coerced: only the literal strings 'True' and 'False' (case-insensitive)
are accepted. Any other value is treated as None (unanswered).
"""
value = value.strip().replace('"', "")
parsed_value = None

if value != "-1":
parsed_value = value
# Determine expected type for this field
field_type = self._target_fields.get(field) if isinstance(self._target_fields, dict) else str
if not isinstance(field_type, type):
field_type = str

if field_type is bool:
# Strictly enforce True/False — no fuzzy matching
if value.lower() == "true":
parsed_value = True
elif value.lower() == "false":
parsed_value = False
else:
print(f"[WARN]: Boolean field '{field}' received unexpected value '{value}'. Defaulting to None.")
parsed_value = None
else:
if value != "-1":
parsed_value = value

if ";" in value:
parsed_value = self.handle_plural_values(value)
Comment thread
Dotify71 marked this conversation as resolved.
Outdated
if field in self._json.keys():
self._json[field].append(parsed_value)
else:
Expand Down
68 changes: 58 additions & 10 deletions src/main.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,6 @@
from typing import Union
Comment thread
Dotify71 marked this conversation as resolved.
Outdated
import os

os.environ["CUDA_VISIBLE_DEVICES"] = ""

# Monkey patch rfdetr to force CPU usage on Mac Silicon / Docker
Expand All @@ -12,22 +14,68 @@
except ImportError:
pass

from commonforms import prepare_form
from commonforms import prepare_form
Comment thread
Dotify71 marked this conversation as resolved.
Outdated
from pypdf import PdfReader
from controller import Controller

def input_fields(num_fields: int):
fields = []
for i in range(num_fields):
field = input(f"Enter description for field {i + 1}: ")
fields.append(field)
return fields

def run_pdf_fill_process(user_input: str, definitions: list, pdf_form_path: Union[str, os.PathLike]):
"""
This function is called by the frontend server.
It receives the raw data, runs the PDF filling logic,
and returns the path to the newly created file.
"""

print("[1] Received request from frontend.")
print(f"[2] PDF template path: {pdf_form_path}")

# Normalize Path/PathLike to a plain string for downstream code
pdf_form_path = os.fspath(pdf_form_path)

if not os.path.exists(pdf_form_path):
print(f"Error: PDF template not found at {pdf_form_path}")
return None # Or raise an exception

print("[3] Starting extraction and PDF filling process...")
try:
controller = Controller()
output_name = controller.fill_form(
user_input=user_input,
fields=definitions,
pdf_form_path=pdf_form_path
)

print("\n----------------------------------")
print(f"✅ Process Complete.")

Check failure on line 55 in src/main.py

View workflow job for this annotation

GitHub Actions / lint

ruff (F541)

src/main.py:55:15: F541 f-string without any placeholders help: Remove extraneous `f` prefix
print(f"Output saved to: {output_name}")

return output_name

except Exception as e:
print(f"An error occurred during PDF generation: {e}")
# Re-raise the exception so the frontend can handle it
raise e
if __name__ == "__main__":
file = "./src/inputs/file.pdf"
user_input = "Hi. The employee's name is John Doe. His job title is managing director. His department supervisor is Jane Doe. His phone number is 123456. His email is jdoe@ucsc.edu. The signature is <Mamañema>, and the date is 01/02/2005"
fields = [
"Employee's name",
"Employee's job title",
"Employee's department supervisor",
"Employee's phone number",
"Employee's email",
"Signature",
"Date",
]
# Fields dict maps each field name to its expected Python type.
# Use `bool` for checkbox/radio fields so the LLM is instructed to
# return exactly True or False instead of fuzzy strings like "yes".
fields = {
"Employee's name": str,
"Employee's job title": str,
"Employee's department supervisor": str,
"Employee's phone number": str,
"Employee's email": str,
"Signature": str,
"Date": str,
}
prepared_pdf = "temp_outfile.pdf"
prepare_form(file, prepared_pdf)

Expand Down
Loading