Create a specific folder for results
This commit is contained in:
parent
14c532b138
commit
3f91fa3e27
4 changed files with 71 additions and 17 deletions
1
.gitignore
vendored
1
.gitignore
vendored
|
|
@ -1,2 +1,3 @@
|
||||||
.idea
|
.idea
|
||||||
document_output
|
document_output
|
||||||
|
results
|
||||||
|
|
|
||||||
10
README.md
10
README.md
|
|
@ -1 +1,9 @@
|
||||||
python analyzer.py --image document.pdf --page 8 && python visualizer.py --doctags output.doctags.txt --pdf document.pdf --page 8 --adjust --show
|
# DocTags Analyzer and Visualizer
|
||||||
|
|
||||||
|
Simple command to process PDF pages with DocTags:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python analyzer.py --image document.pdf --page 8 && python visualizer.py --doctags results/output.doctags.txt --pdf document.pdf --page 8 --adjust --show
|
||||||
|
```
|
||||||
|
|
||||||
|
All output files will be automatically stored in the `results` folder.
|
||||||
33
analyzer.py
33
analyzer.py
|
|
@ -26,14 +26,25 @@ from docling_core.types.doc import ImageRefMode
|
||||||
from docling_core.types.doc.document import DocTagsDocument, DoclingDocument
|
from docling_core.types.doc.document import DocTagsDocument, DoclingDocument
|
||||||
|
|
||||||
|
|
||||||
|
def ensure_results_folder():
|
||||||
|
"""Create the results folder if it doesn't exist."""
|
||||||
|
results_dir = Path("results")
|
||||||
|
if not results_dir.exists():
|
||||||
|
results_dir.mkdir()
|
||||||
|
print(f"Created results directory: {results_dir}")
|
||||||
|
return results_dir
|
||||||
|
|
||||||
|
|
||||||
def parse_arguments():
|
def parse_arguments():
|
||||||
"""Parse command line arguments."""
|
"""Parse command line arguments."""
|
||||||
|
results_dir = ensure_results_folder()
|
||||||
|
|
||||||
parser = argparse.ArgumentParser(description='Convert an image or PDF to docling format')
|
parser = argparse.ArgumentParser(description='Convert an image or PDF to docling format')
|
||||||
parser.add_argument('--image', '-i', type=str, required=True,
|
parser.add_argument('--image', '-i', type=str, required=True,
|
||||||
help='Path to local image file, PDF file, or URL')
|
help='Path to local image file, PDF file, or URL')
|
||||||
parser.add_argument('--prompt', '-p', type=str, default="Convert this page to docling.",
|
parser.add_argument('--prompt', '-p', type=str, default="Convert this page to docling.",
|
||||||
help='Prompt for the model')
|
help='Prompt for the model')
|
||||||
parser.add_argument('--output', '-o', type=str, default="./output.html",
|
parser.add_argument('--output', '-o', type=str, default=str(results_dir / "output.html"),
|
||||||
help='Output file path')
|
help='Output file path')
|
||||||
parser.add_argument('--show', '-s', action='store_true',
|
parser.add_argument('--show', '-s', action='store_true',
|
||||||
help='Show output in browser')
|
help='Show output in browser')
|
||||||
|
|
@ -410,6 +421,9 @@ def process_page(args, model, processor, config, image_path, pil_image, page_num
|
||||||
from mlx_vlm.prompt_utils import apply_chat_template
|
from mlx_vlm.prompt_utils import apply_chat_template
|
||||||
from mlx_vlm.utils import stream_generate
|
from mlx_vlm.utils import stream_generate
|
||||||
|
|
||||||
|
# Ensure results folder exists
|
||||||
|
results_dir = ensure_results_folder()
|
||||||
|
|
||||||
# Prepare input
|
# Prepare input
|
||||||
prompt = args.prompt
|
prompt = args.prompt
|
||||||
output_base = Path(args.output)
|
output_base = Path(args.output)
|
||||||
|
|
@ -419,7 +433,7 @@ def process_page(args, model, processor, config, image_path, pil_image, page_num
|
||||||
if Path(image_path).suffix.lower() == '.pdf' and page_num > 1:
|
if Path(image_path).suffix.lower() == '.pdf' and page_num > 1:
|
||||||
# Get base filename without extension
|
# Get base filename without extension
|
||||||
base_name = output_base.stem
|
base_name = output_base.stem
|
||||||
output_path = output_base.parent / f"{base_name}_page{page_num}{output_base.suffix}"
|
output_path = results_dir / f"{base_name}_page{page_num}{output_base.suffix}"
|
||||||
|
|
||||||
print(f"Processing page {page_num}, output will be saved to {output_path}")
|
print(f"Processing page {page_num}, output will be saved to {output_path}")
|
||||||
|
|
||||||
|
|
@ -460,15 +474,15 @@ def process_page(args, model, processor, config, image_path, pil_image, page_num
|
||||||
os.unlink(temp_img_path)
|
os.unlink(temp_img_path)
|
||||||
print(f"Removed temporary image file")
|
print(f"Removed temporary image file")
|
||||||
|
|
||||||
# Save the raw DocTags to a txt file
|
# Save the raw DocTags to a txt file in results folder
|
||||||
doctags_path = output_path.with_suffix('.doctags.txt')
|
doctags_path = results_dir / f"{output_path.stem}.doctags.txt"
|
||||||
with open(doctags_path, 'w', encoding='utf-8') as f:
|
with open(doctags_path, 'w', encoding='utf-8') as f:
|
||||||
f.write(output)
|
f.write(output)
|
||||||
print(f"Raw DocTags saved to: {doctags_path}")
|
print(f"Raw DocTags saved to: {doctags_path}")
|
||||||
|
|
||||||
# Save the tag analysis if in DocTags-only mode
|
# Save the tag analysis if in DocTags-only mode
|
||||||
if args.doctags_only:
|
if args.doctags_only:
|
||||||
tags_path = output_path.with_suffix('.tags.md')
|
tags_path = results_dir / f"{output_path.stem}.tags.md"
|
||||||
with open(tags_path, 'w', encoding='utf-8') as f:
|
with open(tags_path, 'w', encoding='utf-8') as f:
|
||||||
f.write(tags_analysis)
|
f.write(tags_analysis)
|
||||||
print(f"DocTags analysis saved to: {tags_path}")
|
print(f"DocTags analysis saved to: {tags_path}")
|
||||||
|
|
@ -477,14 +491,14 @@ def process_page(args, model, processor, config, image_path, pil_image, page_num
|
||||||
if not args.doctags_only:
|
if not args.doctags_only:
|
||||||
# Create markdown document
|
# Create markdown document
|
||||||
md_content = create_markdown_document(image_path, output, args)
|
md_content = create_markdown_document(image_path, output, args)
|
||||||
md_path = output_path.with_suffix('.md')
|
md_path = results_dir / f"{output_path.stem}.md"
|
||||||
with open(md_path, 'w', encoding='utf-8') as f:
|
with open(md_path, 'w', encoding='utf-8') as f:
|
||||||
f.write(md_content)
|
f.write(md_content)
|
||||||
print(f"Markdown saved to: {md_path}")
|
print(f"Markdown saved to: {md_path}")
|
||||||
|
|
||||||
# Create HTML document
|
# Create HTML document
|
||||||
html_content = create_html_document(image_path, output, pil_image, args)
|
html_content = create_html_document(image_path, output, pil_image, args)
|
||||||
html_path = output_path.with_suffix('.html')
|
html_path = results_dir / f"{output_path.stem}.html"
|
||||||
with open(html_path, 'w', encoding='utf-8') as f:
|
with open(html_path, 'w', encoding='utf-8') as f:
|
||||||
f.write(html_content)
|
f.write(html_content)
|
||||||
print(f"HTML saved to: {html_path}")
|
print(f"HTML saved to: {html_path}")
|
||||||
|
|
@ -500,7 +514,7 @@ def process_page(args, model, processor, config, image_path, pil_image, page_num
|
||||||
|
|
||||||
# Save a copy of the processed image for reference if in debug mode
|
# Save a copy of the processed image for reference if in debug mode
|
||||||
if args.debug:
|
if args.debug:
|
||||||
img_debug_path = output_path.with_suffix('.debug.png')
|
img_debug_path = results_dir / f"{output_path.stem}.debug.png"
|
||||||
pil_image.save(img_debug_path)
|
pil_image.save(img_debug_path)
|
||||||
print(f"Saved debug image to: {img_debug_path}")
|
print(f"Saved debug image to: {img_debug_path}")
|
||||||
|
|
||||||
|
|
@ -508,6 +522,9 @@ def process_page(args, model, processor, config, image_path, pil_image, page_num
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
|
# Ensure results folder exists
|
||||||
|
ensure_results_folder()
|
||||||
|
|
||||||
# Parse arguments
|
# Parse arguments
|
||||||
args = parse_arguments()
|
args = parse_arguments()
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -9,7 +9,7 @@ Usage:
|
||||||
import argparse
|
import argparse
|
||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
import sys # Add missing import
|
import sys
|
||||||
import base64
|
import base64
|
||||||
from io import BytesIO
|
from io import BytesIO
|
||||||
import tempfile
|
import tempfile
|
||||||
|
|
@ -21,8 +21,18 @@ import pdf2image
|
||||||
# Regular expression to extract location data
|
# Regular expression to extract location data
|
||||||
LOC_PATTERN = r'<loc_(\d+)><loc_(\d+)><loc_(\d+)><loc_(\d+)>'
|
LOC_PATTERN = r'<loc_(\d+)><loc_(\d+)><loc_(\d+)><loc_(\d+)>'
|
||||||
|
|
||||||
|
def ensure_results_folder():
|
||||||
|
"""Create the results folder if it doesn't exist."""
|
||||||
|
results_dir = Path("results")
|
||||||
|
if not results_dir.exists():
|
||||||
|
results_dir.mkdir()
|
||||||
|
print(f"Created results directory: {results_dir}")
|
||||||
|
return results_dir
|
||||||
|
|
||||||
def parse_arguments():
|
def parse_arguments():
|
||||||
"""Parse command line arguments."""
|
"""Parse command line arguments."""
|
||||||
|
results_dir = ensure_results_folder()
|
||||||
|
|
||||||
parser = argparse.ArgumentParser(description='Visualize zones identified in DocTags format')
|
parser = argparse.ArgumentParser(description='Visualize zones identified in DocTags format')
|
||||||
parser.add_argument('--doctags', '-d', type=str, required=True,
|
parser.add_argument('--doctags', '-d', type=str, required=True,
|
||||||
help='Path to DocTags file')
|
help='Path to DocTags file')
|
||||||
|
|
@ -30,7 +40,7 @@ def parse_arguments():
|
||||||
help='Path to original PDF file')
|
help='Path to original PDF file')
|
||||||
parser.add_argument('--page', type=int, default=8,
|
parser.add_argument('--page', type=int, default=8,
|
||||||
help='Page number in PDF (starts at 1, default: 8)')
|
help='Page number in PDF (starts at 1, default: 8)')
|
||||||
parser.add_argument('--output', '-o', type=str, default="visualization.html",
|
parser.add_argument('--output', '-o', type=str, default=str(results_dir / "visualization.html"),
|
||||||
help='Output HTML file path')
|
help='Output HTML file path')
|
||||||
parser.add_argument('--dpi', type=int, default=200,
|
parser.add_argument('--dpi', type=int, default=200,
|
||||||
help='DPI for PDF rendering')
|
help='DPI for PDF rendering')
|
||||||
|
|
@ -517,14 +527,14 @@ def create_debug_image(image, zones, page_num, output_path):
|
||||||
|
|
||||||
return debug_img
|
return debug_img
|
||||||
|
|
||||||
# Improved auto-adjustment function for visualizer.py
|
|
||||||
# Replace this function in your visualizer.py script
|
|
||||||
|
|
||||||
def process_page(pdf_path, page_num, doctags_path, output_base, dpi=200, show=False, scale=1.0, scale_x=None, scale_y=None, adjust=True):
|
def process_page(pdf_path, page_num, doctags_path, output_base, dpi=200, show=False, scale=1.0, scale_x=None, scale_y=None, adjust=True):
|
||||||
"""Process a single page of the PDF with visualization."""
|
"""Process a single page of the PDF with visualization."""
|
||||||
|
# Ensure results folder exists
|
||||||
|
results_dir = ensure_results_folder()
|
||||||
|
|
||||||
# Generate output paths for this page
|
# Generate output paths for this page
|
||||||
output_name = f"{Path(output_base).stem}_page_{page_num}{Path(output_base).suffix}"
|
output_name = f"{Path(output_base).stem}_page_{page_num}{Path(output_base).suffix}"
|
||||||
output_path = Path(output_base).parent / output_name
|
output_path = results_dir / output_name
|
||||||
debug_output = output_path.with_suffix('.debug.png')
|
debug_output = output_path.with_suffix('.debug.png')
|
||||||
|
|
||||||
# Load the page image
|
# Load the page image
|
||||||
|
|
@ -613,6 +623,9 @@ def process_page(pdf_path, page_num, doctags_path, output_base, dpi=200, show=Fa
|
||||||
|
|
||||||
def process_all_pages(pdf_path, doctags_path, output_base, dpi=200, show_last=False, scale=1.0, scale_x=None, scale_y=None, adjust=False):
|
def process_all_pages(pdf_path, doctags_path, output_base, dpi=200, show_last=False, scale=1.0, scale_x=None, scale_y=None, adjust=False):
|
||||||
"""Process all pages of the PDF and create visualizations."""
|
"""Process all pages of the PDF and create visualizations."""
|
||||||
|
# Ensure results folder exists
|
||||||
|
results_dir = ensure_results_folder()
|
||||||
|
|
||||||
# Get total page count
|
# Get total page count
|
||||||
total_pages = count_pdf_pages(pdf_path)
|
total_pages = count_pdf_pages(pdf_path)
|
||||||
if total_pages == 0:
|
if total_pages == 0:
|
||||||
|
|
@ -629,7 +642,7 @@ def process_all_pages(pdf_path, doctags_path, output_base, dpi=200, show_last=Fa
|
||||||
|
|
||||||
# Generate output paths for this page
|
# Generate output paths for this page
|
||||||
output_name = f"{Path(output_base).stem}_page_{page_num}{Path(output_base).suffix}"
|
output_name = f"{Path(output_base).stem}_page_{page_num}{Path(output_base).suffix}"
|
||||||
output_path = Path(output_base).parent / output_name
|
output_path = results_dir / output_name
|
||||||
|
|
||||||
# Process the page
|
# Process the page
|
||||||
success = process_page(pdf_path, page_num, doctags_path, output_path, dpi, False, scale, scale_x, scale_y, adjust)
|
success = process_page(pdf_path, page_num, doctags_path, output_path, dpi, False, scale, scale_x, scale_y, adjust)
|
||||||
|
|
@ -645,6 +658,9 @@ def process_all_pages(pdf_path, doctags_path, output_base, dpi=200, show_last=Fa
|
||||||
return True
|
return True
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
|
# Ensure results folder exists
|
||||||
|
ensure_results_folder()
|
||||||
|
|
||||||
# Parse arguments
|
# Parse arguments
|
||||||
args = parse_arguments()
|
args = parse_arguments()
|
||||||
|
|
||||||
|
|
@ -711,6 +727,16 @@ def generate_doctags_from_pdf(input_pdf, page_num, output_doctags, prompt="Conve
|
||||||
Returns:
|
Returns:
|
||||||
bool: True if successful, False otherwise
|
bool: True if successful, False otherwise
|
||||||
"""
|
"""
|
||||||
|
# Ensure results folder exists
|
||||||
|
results_dir = ensure_results_folder()
|
||||||
|
|
||||||
|
# Update output path to be in results folder
|
||||||
|
if not isinstance(output_doctags, Path):
|
||||||
|
output_doctags = Path(output_doctags)
|
||||||
|
|
||||||
|
if output_doctags.parent != results_dir:
|
||||||
|
output_doctags = results_dir / output_doctags.name
|
||||||
|
|
||||||
import subprocess
|
import subprocess
|
||||||
import sys
|
import sys
|
||||||
|
|
||||||
|
|
@ -773,12 +799,14 @@ def generate_doctags_from_pdf(input_pdf, page_num, output_doctags, prompt="Conve
|
||||||
# Command-line interface for the DocTags generator
|
# Command-line interface for the DocTags generator
|
||||||
def generate_doctags_cli():
|
def generate_doctags_cli():
|
||||||
"""Command-line interface for generating DocTags."""
|
"""Command-line interface for generating DocTags."""
|
||||||
|
results_dir = ensure_results_folder()
|
||||||
|
|
||||||
parser = argparse.ArgumentParser(description='Generate DocTags from a PDF page')
|
parser = argparse.ArgumentParser(description='Generate DocTags from a PDF page')
|
||||||
parser.add_argument('--pdf', type=str, required=True,
|
parser.add_argument('--pdf', type=str, required=True,
|
||||||
help='Path to PDF file')
|
help='Path to PDF file')
|
||||||
parser.add_argument('--page', type=int, default=1,
|
parser.add_argument('--page', type=int, default=1,
|
||||||
help='Page number (starts at 1)')
|
help='Page number (starts at 1)')
|
||||||
parser.add_argument('--output', type=str, default='output.doctags.txt',
|
parser.add_argument('--output', type=str, default=str(results_dir / 'output.doctags.txt'),
|
||||||
help='Output DocTags file')
|
help='Output DocTags file')
|
||||||
parser.add_argument('--prompt', type=str, default='Convert this page to docling.',
|
parser.add_argument('--prompt', type=str, default='Convert this page to docling.',
|
||||||
help='Prompt for the model')
|
help='Prompt for the model')
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue