import zlib
import re

def parse_pdf_streams(pdf_path):
    with open(pdf_path, 'rb') as f:
        data = f.read()

    streams = re.findall(b'stream[\r\n]+(.*?)[\r\n]+endstream', data, re.DOTALL)
    print(f"=== {os.path.basename(pdf_path)} (Found {len(streams)} streams) ===")
    
    text_pieces = []
    for s in streams:
        try:
            decomp = zlib.decompress(s)
            # Find text inside BT ... ET
            bt_matches = re.findall(b'BT(.*?)ET', decomp, re.DOTALL)
            for bt in bt_matches:
                # Find (text) Tj or [(text)] TJ
                tjs = re.findall(b'\((.*?)\)\s*Tj', bt)
                for t in tjs:
                    try:
                        text_pieces.append(t.decode('latin1'))
                    except:
                        pass
                tj_arrays = re.findall(b'\[(.*?)\]\s*TJ', bt)
                for tja in tj_arrays:
                    inner_texts = re.findall(b'\((.*?)\)', tja)
                    line = "".join([it.decode('latin1', errors='ignore') for it in inner_texts])
                    if len(line.strip()) > 2:
                        text_pieces.append(line)
        except Exception as e:
            pass
            
    print("Extracted text (first 500 chars):")
    joined = " \n ".join([t.strip() for t in text_pieces if t.strip()])
    print(joined[:600])
    print("\n" + "="*50 + "\n")

import os
pdf_dir = r"d:\iss\Franciscan Society\assets\pdf"
for f in os.listdir(pdf_dir):
    if f.endswith('.pdf'):
        parse_pdf_streams(os.path.join(pdf_dir, f))
