Files
Tracker/Code/Tools/Samples/generate_samples.py
T
2026-02-21 21:55:42 +00:00

554 lines
20 KiB
Python

#!/usr/bin/env python3
"""
JingTian-Tracker Sample Document Generator
Generates 7 realistic sample documents for testing the JingTian document
processing pipeline. Uses pre-generated content pools (no LLM needed at runtime).
Usage:
python generate_samples.py --output ./outputs
python generate_samples.py --output ./outputs --seed 42
Documents generated:
1. DOCX - Letter TO client (Client/{name}/)
2. PDF - Scanned letter FROM client (Client/{name}/)
3. PNG - Email screenshot FROM client (Client/{name}/)
4. PDF - IPD filing receipt (Admin/IPD e-filing/)
5. DOCX - Internal memo (Admin/General Matter/)
6. XLSX - Invoice schedule (Billing/Draft Bills/)
7. PDF - TM registry record (IP/)
+ Copies 2 real PDFs to IP/
"""
import argparse
import json
import random
import shutil
import sys
from datetime import datetime, timedelta
from pathlib import Path
# Document generators
from generators.docx_letter_to_client import generate as gen_letter_to
from generators.pdf_scanned_letter import generate as gen_scanned_letter
from generators.image_email import generate as gen_email_image
from generators.pdf_filing_receipt import generate as gen_filing_receipt
from generators.docx_memo import generate as gen_memo
from generators.xlsx_invoice import generate as gen_invoice
from generators.pdf_tm_record import generate as gen_tm_record
DATA_DIR = Path(__file__).parent / "data"
REAL_PDFS_DIR = Path(__file__).parent / "real_pdfs"
def load_pools():
"""Load all data pools."""
pools = {}
for name in ["clients", "trademarks", "names", "content_pools"]:
path = DATA_DIR / f"{name}.json"
if not path.exists():
print(f"Error: {path} not found. Run generate_content_pools.py first.")
sys.exit(1)
pools[name] = json.loads(path.read_text(encoding="utf-8"))
return pools
def random_future_date(min_days=30, max_days=180):
"""Generate a random date in the future."""
delta = timedelta(days=random.randint(min_days, max_days))
return (datetime.now() + delta).strftime("%d-%m-%Y")
def random_tm_number():
"""Generate a realistic 9-digit TM number."""
prefix = random.choice(["306", "307"])
suffix = str(random.randint(100000, 999999))
return prefix + suffix
def sanitize_dirname(name):
"""Sanitize a string for use as a directory/file name."""
# Replace dots at end (Windows issue), replace spaces with underscores
name = name.replace(" ", "_")
name = name.rstrip(".")
# Remove other problematic chars
for ch in ["<", ">", ":", '"', "/", "\\", "|", "?", "*"]:
name = name.replace(ch, "")
return name
def pick_client(pools):
"""Pick a random client with contact info."""
client = random.choice(pools["clients"]["clients"])
return client
def pick_trademark(pools):
"""Pick a random trademark from the real data."""
tm = random.choice(pools["trademarks"]["trademarks"])
return tm
def get_body(item):
"""Extract body text from a content pool item (str or dict with 'body' key)."""
if isinstance(item, dict):
return item.get("body", str(item))
return str(item)
def fill_template(template, replacements):
"""Fill placeholders in a template string."""
result = get_body(template) if not isinstance(template, str) else template
for key, val in replacements.items():
result = result.replace(f"{{{key}}}", str(val))
return result
def generate_all(output_dir, pools):
"""Generate all 7 documents and copy real PDFs."""
manifest = {
"generated_at": datetime.now().isoformat(),
"documents": [],
}
firm = pools["names"]["firm"]
ipd = pools["names"]["ipd"]
content = pools["content_pools"]
# ── 1. DOCX: Letter TO client ──────────────────────────────────
client1 = pick_client(pools)
tm1 = pick_trademark(pools)
deadline1 = random_future_date(30, 120)
body1 = random.choice(content["letter_to_client_bodies"])
closing1 = random.choice(content["letter_closings"])
replacements1 = {
"deadline": deadline1,
"tm_number": tm1["number"],
"tm_text": tm1["text"],
"client_name": client1["name"],
"contact_person": client1["contact_person"],
}
client_dir = output_dir / "Client" / sanitize_dirname(client1["name"])
client_dir.mkdir(parents=True, exist_ok=True)
fname1 = f"Letter_Re_TM{tm1['number']}.docx"
gen_letter_to(
output_path=client_dir / fname1,
firm=firm,
client=client1,
body=fill_template(body1, replacements1),
closing=fill_template(closing1, replacements1),
ref_number=f"JT/{datetime.now().year}/{random.randint(1000, 9999)}",
date=datetime.now().strftime("%d %B %Y"),
re_line=f"Trademark Application No. {tm1['number']} - {tm1['text']}",
)
manifest["documents"].append(
{
"filename": fname1,
"path": f"Client/{sanitize_dirname(client1['name'])}/{fname1}",
"type": "Letter to Client",
"format": "docx",
"expected_extraction": {
"document_type": "Client Correspondence",
"deadline": deadline1,
"client": client1["name"],
"tm_number": tm1["number"],
},
}
)
print(f" [1/7] DOCX letter to client: {fname1}")
# ── 2. PDF: Scanned letter FROM client ─────────────────────────
client2 = pick_client(pools)
tm2 = pick_trademark(pools)
deadline2 = random_future_date(14, 90)
body2 = random.choice(content["letter_from_client_bodies"])
has_deadline2 = "{deadline}" in body2
replacements2 = {
"deadline": deadline2,
"tm_number": tm2["number"],
"tm_text": tm2["text"],
"client_name": client2["name"],
"contact_person": client2["contact_person"],
}
client2_dir = output_dir / "Client" / sanitize_dirname(client2["name"])
client2_dir.mkdir(parents=True, exist_ok=True)
fname2 = f"Client_Instructions_{sanitize_dirname(client2['name'])}.pdf"
gen_scanned_letter(
output_path=client2_dir / fname2,
from_name=client2["contact_person"],
from_company=client2["name"],
from_address=client2["address"],
to_name=firm["attorney"]["name"],
to_firm=firm["name"],
body=fill_template(body2, replacements2),
date=datetime.now().strftime("%d %B %Y"),
)
manifest["documents"].append(
{
"filename": fname2,
"path": f"Client/{sanitize_dirname(client2['name'])}/{fname2}",
"type": "Letter from Client (Scanned)",
"format": "pdf_scanned",
"ocr_required": True,
"expected_extraction": {
"document_type": "Client Instructions",
"deadline": deadline2 if has_deadline2 else None,
"client": client2["name"],
"tm_number": tm2["number"] if "{tm_number}" in body2 else None,
},
}
)
print(f" [2/7] PDF scanned letter from client: {fname2}")
# ── 3. PNG: Email FROM client ──────────────────────────────────
client3 = pick_client(pools)
tm3 = pick_trademark(pools)
deadline3 = random_future_date(7, 60)
email_body = random.choice(content["email_from_client_bodies"])
email_subject = random.choice(content["email_subjects"])
has_deadline3 = "{deadline}" in email_body
replacements3 = {
"deadline": deadline3,
"tm_number": tm3["number"],
"tm_text": tm3["text"],
"client_name": client3["name"],
"contact_person": client3["contact_person"],
}
client3_dir = output_dir / "Client" / sanitize_dirname(client3["name"])
client3_dir.mkdir(parents=True, exist_ok=True)
fname3 = f"Email_{sanitize_dirname(client3['name'])}_{datetime.now().strftime('%Y%m%d')}.png"
gen_email_image(
output_path=client3_dir / fname3,
from_email=client3.get(
"email", f"info@{client3['name'].lower().replace(' ', '')}.com"
),
from_name=client3["contact_person"],
to_email="benjamin.choi@jingtian.com",
to_name=firm["attorney"]["name"],
subject=fill_template(email_subject, replacements3),
body=fill_template(email_body, replacements3),
date=datetime.now().strftime("%A, %d %B %Y %H:%M"),
)
manifest["documents"].append(
{
"filename": fname3,
"path": f"Client/{sanitize_dirname(client3['name'])}/{fname3}",
"type": "Email from Client (Screenshot)",
"format": "png",
"ocr_required": True,
"expected_extraction": {
"document_type": "Client Email",
"deadline": deadline3 if has_deadline3 else None,
"client": client3["name"],
"tm_number": tm3["number"]
if "{tm_number}" in email_body or "{tm_number}" in email_subject
else None,
},
}
)
print(f" [3/7] PNG email screenshot: {fname3}")
# ── 4. PDF: IPD Filing Receipt ─────────────────────────────────
client4 = pick_client(pools)
tm_number4 = random_tm_number()
tm_text4 = random.choice(
[
client4["name"].split()[0].upper(),
random.choice(["NOVA", "APEX", "STELLAR", "ZENITH", "PRIMEX", "VANTAGE"]),
]
)
filing_date = datetime.now().strftime("%d-%m-%Y")
response_deadline4 = random_future_date(60, 120)
nice_class = random.choice(list(pools["trademarks"]["nice_classes"].keys()))
filing_dir = output_dir / "Admin" / "IPD e-filing"
filing_dir.mkdir(parents=True, exist_ok=True)
fname4 = f"Filing_Receipt_{tm_number4}.pdf"
gen_filing_receipt(
output_path=filing_dir / fname4,
ipd=ipd,
tm_number=tm_number4,
tm_text=tm_text4,
applicant=client4["name"],
applicant_address=client4["address"],
agent=firm["name"],
agent_address=firm["address"],
filing_date=filing_date,
response_deadline=response_deadline4,
nice_class=nice_class,
class_description=pools["trademarks"]["nice_classes"][nice_class],
)
manifest["documents"].append(
{
"filename": fname4,
"path": f"Admin/IPD e-filing/{fname4}",
"type": "IPD Filing Receipt",
"format": "pdf_native",
"expected_extraction": {
"document_type": "Filing Receipt",
"deadline": response_deadline4,
"client": client4["name"],
"tm_number": tm_number4,
},
}
)
print(f" [4/7] PDF filing receipt: {fname4}")
# ── 5. DOCX: Internal Memo ─────────────────────────────────────
memo_body = random.choice(content["memo_bodies"])
clients_for_memo = random.sample(
pools["clients"]["clients"], min(3, len(pools["clients"]["clients"]))
)
tms_for_memo = random.sample(
pools["trademarks"]["trademarks"],
min(3, len(pools["trademarks"]["trademarks"])),
)
memo_replacements = {}
memo_deadlines = []
nice_classes_list = list(pools["trademarks"]["nice_classes"].keys())
for i in range(3):
dl = random_future_date(14 + i * 30, 60 + i * 60)
memo_deadlines.append(dl)
memo_replacements[f"tm_number_{i + 1}"] = (
tms_for_memo[i]["number"] if i < len(tms_for_memo) else random_tm_number()
)
# Use TM number as the mark reference (Chinese text causes rendering issues in memos)
tm_text = tms_for_memo[i]["text"] if i < len(tms_for_memo) else "N/A"
tm_num = (
tms_for_memo[i]["number"] if i < len(tms_for_memo) else random_tm_number()
)
# If text is CJK, show as "No. XXXXXXX (text)", otherwise just the text
memo_replacements[f"tm_text_{i + 1}"] = (
tm_text if tm_text.isascii() else f"No. {tm_num}"
)
memo_replacements[f"deadline_{i + 1}"] = dl
memo_replacements[f"client_{i + 1}"] = (
clients_for_memo[i]["name"] if i < len(clients_for_memo) else "Various"
)
memo_replacements[f"class_{i + 1}"] = random.choice(nice_classes_list)
memo_replacements[f"contact_person_{i + 1}"] = (
clients_for_memo[i]["contact_person"]
if i < len(clients_for_memo)
else "N/A"
)
# Also fill generic placeholders
memo_replacements["deadline"] = memo_deadlines[0]
memo_replacements["contact_person"] = clients_for_memo[0]["contact_person"]
memo_replacements["client_name"] = clients_for_memo[0]["name"]
memo_dir = output_dir / "Admin" / "General Matter"
memo_dir.mkdir(parents=True, exist_ok=True)
fname5 = f"Memo_{datetime.now().strftime('%Y%m%d')}_{random.randint(100, 999)}.docx"
gen_memo(
output_path=memo_dir / fname5,
firm=firm,
body=fill_template(memo_body, memo_replacements),
date=datetime.now().strftime("%d %B %Y"),
subject="Upcoming Trademark Deadlines - Action Required",
)
manifest["documents"].append(
{
"filename": fname5,
"path": f"Admin/General Matter/{fname5}",
"type": "Internal Memo",
"format": "docx",
"expected_extraction": {
"document_type": "Internal Memo",
"deadlines": memo_deadlines,
"tm_numbers": [
r.get(f"tm_number_{i + 1}")
for i, r in enumerate([memo_replacements] * 3)
],
},
}
)
print(f" [5/7] DOCX internal memo: {fname5}")
# ── 6. XLSX: Invoice Schedule ──────────────────────────────────
invoice_clients = random.sample(
pools["clients"]["clients"], min(6, len(pools["clients"]["clients"]))
)
invoice_tms = random.sample(
pools["trademarks"]["trademarks"],
min(6, len(pools["trademarks"]["trademarks"])),
)
invoice_descs = content["invoice_descriptions"]
invoice_rows = []
for i in range(min(6, len(invoice_clients))):
cls = random.choice(list(pools["trademarks"]["nice_classes"].keys()))
desc = fill_template(random.choice(invoice_descs), {"class": cls})
due = random_future_date(14, 90)
amount = random.randint(5, 150) * 1000
invoice_rows.append(
{
"client": invoice_clients[i]["name"],
"matter_ref": f"JT/{datetime.now().year}/{random.randint(1000, 9999)}",
"tm_number": invoice_tms[i]["number"]
if i < len(invoice_tms)
else random_tm_number(),
"description": desc,
"amount_hkd": amount,
"due_date": due,
"status": random.choice(["Draft", "Sent", "Overdue", "Paid"]),
}
)
billing_dir = output_dir / "Billing" / "Draft Bills"
billing_dir.mkdir(parents=True, exist_ok=True)
fname6 = f"Invoice_Schedule_{datetime.now().strftime('%Y')}Q{(datetime.now().month - 1) // 3 + 1}.xlsx"
gen_invoice(
output_path=billing_dir / fname6,
rows=invoice_rows,
firm=firm,
)
manifest["documents"].append(
{
"filename": fname6,
"path": f"Billing/Draft Bills/{fname6}",
"type": "Invoice Schedule",
"format": "xlsx",
"expected_extraction": {
"document_type": "Invoice Schedule",
"deadlines": [r["due_date"] for r in invoice_rows],
"clients": [r["client"] for r in invoice_rows],
},
}
)
print(f" [6/7] XLSX invoice schedule: {fname6}")
# ── 7. PDF: TM Registry Record ─────────────────────────────────
tm7 = pick_trademark(pools)
client7 = pick_client(pools)
status_type = random.choice(pools["trademarks"]["statuses"])
# Some statuses imply deadlines
has_implicit_deadline = status_type in [
"Application Opposed",
"Examined - First Examination Report Issued",
"Examined - Further Examination Report Issued",
]
ip_dir = output_dir / "IP"
ip_dir.mkdir(parents=True, exist_ok=True)
fname7 = f"TM_Record_{tm7['number']}.pdf"
gen_tm_record(
output_path=ip_dir / fname7,
ipd=ipd,
tm_number=tm7["number"],
tm_text=tm7["text"],
status=status_type,
nice_class=str(tm7["classes"][0]) if tm7["classes"] else "5",
class_description=pools["trademarks"]["nice_classes"].get(
str(tm7["classes"][0]), "General goods and services"
),
applicant=client7["name"],
applicant_address=client7["address"],
agent=firm["name"],
agent_address=firm["address"],
filing_date=tm7.get("filing_date", datetime.now().strftime("%d-%m-%Y")),
publication_date=(
datetime.now() - timedelta(days=random.randint(30, 180))
).strftime("%d-%m-%Y"),
)
manifest["documents"].append(
{
"filename": fname7,
"path": f"IP/{fname7}",
"type": "TM Registry Record",
"format": "pdf_native",
"expected_extraction": {
"document_type": "Trademark Registry Record",
"deadline": None,
"deadline_note": "Implicit deadline based on status"
if has_implicit_deadline
else "No deadline",
"client": client7["name"],
"tm_number": tm7["number"],
"status": status_type,
},
}
)
print(f" [7/7] PDF TM registry record: {fname7}")
# ── Copy real PDFs ─────────────────────────────────────────────
if REAL_PDFS_DIR.exists():
for pdf in REAL_PDFS_DIR.glob("*.pdf"):
dest = ip_dir / pdf.name
shutil.copy2(pdf, dest)
manifest["documents"].append(
{
"filename": pdf.name,
"path": f"IP/{pdf.name}",
"type": "Real Document (not generated)",
"format": "pdf",
"expected_extraction": None,
}
)
print(f" [+] Copied real PDF: {pdf.name}")
# ── Save manifest ──────────────────────────────────────────────
llm_dir = output_dir / "_LLM"
llm_dir.mkdir(parents=True, exist_ok=True)
manifest_path = llm_dir / "manifest.json"
manifest_path.write_text(
json.dumps(manifest, indent=2, ensure_ascii=False),
encoding="utf-8",
)
print(f"\n Manifest saved to {manifest_path}")
return manifest
def main():
parser = argparse.ArgumentParser(description="Generate JingTian sample documents")
parser.add_argument(
"--output",
"-o",
default=str(Path(__file__).parent / "outputs"),
help="Output directory (default: ./outputs)",
)
parser.add_argument(
"--seed", "-s", type=int, default=None, help="Random seed for reproducible runs"
)
args = parser.parse_args()
if args.seed is not None:
random.seed(args.seed)
print(f"Using seed: {args.seed}")
output_dir = Path(args.output)
output_dir.mkdir(parents=True, exist_ok=True)
print(f"Generating JingTian sample documents...")
print(f"Output: {output_dir}\n")
pools = load_pools()
manifest = generate_all(output_dir, pools)
print(f"\nDone! Generated {len(manifest['documents'])} documents.")
if __name__ == "__main__":
main()