Initial commit

This commit is contained in:
2025-07-11 13:52:19 +00:00
commit e0c489f625
362 changed files with 27286 additions and 0 deletions
View File
Binary file not shown.
Binary file not shown.
+423
View File
@@ -0,0 +1,423 @@
import os
from modules.logger_tool import initialise_logger
logger = initialise_logger(log_name="pdf", log_level=os.getenv("LOG_LEVEL"), log_dir=os.getenv("LOG_PATH"), log_format="default", runtime=True)
from fastapi import APIRouter, UploadFile, File, HTTPException
from fastapi.responses import JSONResponse
from pathlib import Path
import tempfile
from PIL import Image
import io
import base64
import traceback
import sys
import subprocess
from concurrent.futures import ThreadPoolExecutor, as_completed, TimeoutError
import asyncio
import psutil
import math
import time
from pdfminer.high_level import extract_pages
from pdfminer.layout import LTTextContainer, LTChar, LTLine, LTRect, LTFigure, LTTextBox, LTTextBoxHorizontal, LTTextLine
import re
router = APIRouter()
# Global semaphore to control total concurrent PDF processing
MAX_CONCURRENT_PROCESSING = 4 # Adjust based on server capacity
processing_semaphore = asyncio.Semaphore(MAX_CONCURRENT_PROCESSING)
def calculate_optimal_workers():
"""Calculate optimal number of worker threads based on system resources."""
cpu_count = os.cpu_count() or 4
available_memory = psutil.virtual_memory().available
memory_per_worker = 500 * 1024 * 1024 # 500MB per worker estimate
# Calculate workers based on CPU and memory constraints
cpu_based_workers = max(1, cpu_count - 1) # Leave one core free
memory_based_workers = max(1, int(available_memory / memory_per_worker))
# Take the minimum of CPU and memory-based calculations
optimal_workers = min(cpu_based_workers, memory_based_workers)
# Cap at a reasonable maximum
final_workers = min(optimal_workers, 8) # Maximum 8 workers per process
logger.info("Resource utilization:", {
"total_cpus": cpu_count,
"available_memory_gb": available_memory / (1024**3),
"cpu_based_workers": cpu_based_workers,
"memory_based_workers": memory_based_workers,
"final_workers": final_workers
})
return final_workers
def is_heading(textbox, page_height):
"""Determine if a textbox is likely a heading based on font size and position."""
if not isinstance(textbox, LTTextContainer):
return False, 0
# Get the most common font size in the textbox
font_sizes = []
for text_line in textbox._objs:
if isinstance(text_line, LTTextLine):
font_sizes.extend(
char.size
for char in text_line._objs
if isinstance(char, LTChar)
)
if not font_sizes:
return False, 0
most_common_size = max(set(font_sizes), key=font_sizes.count)
# Position near top of page suggests a heading
is_near_top = textbox.y1 > (page_height - 100)
# Determine heading level based on font size and position
if most_common_size > 20 or is_near_top:
return True, 1
elif most_common_size > 16:
return True, 2
elif most_common_size > 14:
return True, 3
return False, 0
def clean_text(text):
"""Clean and normalize text content."""
# Remove multiple spaces and newlines
text = re.sub(r'\s+', ' ', text)
# Remove special characters often found in PDFs
text = re.sub(r'[^\x00-\x7F]+', '', text)
return text.strip()
def extract_page_text(page):
"""Extract text from a PDF page and format as markdown."""
page_height = page.height
text_elements = []
current_list_items = []
# First pass: collect all text elements and identify their roles
for element in page:
if isinstance(element, LTTextContainer):
text = clean_text(element.get_text())
if not text:
continue
is_head, level = is_heading(element, page_height)
# Check if this looks like a list item
is_list_item = bool(re.match(r'^[\u2022\u2023\u25E6\u2043\u2219•\-*]\s', text))
if is_head:
# If we have pending list items, add them first
if current_list_items:
text_elements.extend(current_list_items)
current_list_items = []
text_elements.append((f"{'#' * level} {text.lstrip('1234567890.-* ')}", element.y1))
elif is_list_item:
current_list_items.append((f"* {text.lstrip('1234567890.-* ')}", element.y1))
else:
# If this is regular text and we have pending list items
if current_list_items:
# Check if this text is part of the same list (similar y-position)
if any(abs(item[1] - element.y1) < 20 for item in current_list_items):
current_list_items.append((f"* {text}", element.y1))
continue
else:
# Add pending list items before adding this text
text_elements.extend(current_list_items)
current_list_items = []
text_elements.append((text, element.y1))
# Add any remaining list items
if current_list_items:
text_elements.extend(current_list_items)
# Sort elements by vertical position (top to bottom)
text_elements.sort(key=lambda x: -x[1])
# Return just the text parts, properly formatted
return '\n\n'.join(element[0] for element in text_elements)
def process_page(temp_dir: str, pdf_path: str, page_info: tuple, timeout: int = 30) -> dict:
"""
Worker function to process a single page and maintain A4 proportions.
Args:
temp_dir: Path to temporary directory
pdf_path: Path to PDF file
page_info: Tuple of (index, page_number)
timeout: Maximum time in seconds to process a single page
Returns:
dict: Processed page information
"""
i, page_idx = page_info
page_num = page_idx + 1 # PDF pages are 1-indexed
output_prefix = str(Path(temp_dir) / f"page_{page_num}")
try:
# Extract text from PDF page
pages = list(extract_pages(pdf_path, page_numbers=[page_idx]))
page_text = extract_page_text(pages[0]) if pages else ""
# Convert PDF page to PNG with timeout
process = subprocess.Popen(
[
'pdftoppm',
'-png',
'-singlefile',
'-f',
str(page_num),
'-l',
str(page_num),
'-r',
'600', # High resolution for better quality
pdf_path,
output_prefix,
],
stdout=subprocess.PIPE,
stderr=subprocess.PIPE
)
try:
stdout, stderr = process.communicate(timeout=timeout)
except subprocess.TimeoutExpired:
process.kill()
raise TimeoutError(f"Page {page_num} processing timed out after {timeout} seconds")
if process.returncode != 0:
raise Exception(f"pdftoppm failed for page {page_num}: {stderr.decode()}")
output_file = f"{output_prefix}.png"
if not Path(output_file).exists():
raise Exception(f"Could not find output file for page {page_num}")
# Open and process the image
with Image.open(output_file) as img:
result = _process_image(img, i)
if result['success']:
result['meta'] = {
'text': page_text,
'format': 'markdown'
}
return result
except Exception as e:
logger.error(f"Error processing page {page_num}: {str(e)}")
return {
"index": i,
"error": str(e),
"success": False,
}
def _process_image(img: Image.Image, index: int) -> dict:
"""Process a single image, maintaining A4 proportions."""
try:
# Determine orientation and target dimensions
is_portrait = img.height > img.width
target_height = 720 # Fixed height to match frontend slide height
if is_portrait:
# A4 portrait ratio is 210:297
target_width = int(target_height * (210/297))
else:
# A4 landscape ratio is 297:210
target_width = int(target_height * (297/210))
# Resize image maintaining aspect ratio
img = img.resize((target_width, target_height), Image.Resampling.LANCZOS)
# Convert to base64
buffered = io.BytesIO()
img.save(buffered, format="PNG", optimize=True)
img_str = base64.b64encode(buffered.getvalue()).decode()
return {
"index": index,
"data": f"data:image/png;base64,{img_str}",
"success": True,
"dimensions": {
"width": target_width,
"height": target_height,
"orientation": "portrait" if is_portrait else "landscape"
}
}
except Exception as e:
logger.error(f"Error processing image for page {index}: {str(e)}")
return {
"index": index,
"error": str(e),
"success": False,
}
async def process_pages_in_chunks(temp_dir: str, pdf_path: str, visible_pages: list, chunk_size: int = 5):
"""Process pages in chunks to manage memory better."""
all_processed_pages = []
num_workers = calculate_optimal_workers()
total_chunks = math.ceil(len(visible_pages) / chunk_size)
logger.info("Starting page processing:", {
"total_pages": len(visible_pages),
"chunk_size": chunk_size,
"total_chunks": total_chunks,
"workers_per_chunk": num_workers
})
# Process pages in chunks
for chunk_index in range(0, len(visible_pages), chunk_size):
chunk = visible_pages[chunk_index:chunk_index + chunk_size]
processed_chunk = []
current_chunk_num = (chunk_index // chunk_size) + 1
logger.info(f"Processing chunk {current_chunk_num}/{total_chunks}", {
"chunk_size": len(chunk),
"chunk_start_index": chunk_index,
"memory_usage_gb": psutil.Process().memory_info().rss / (1024**3)
})
start_time = time.time()
with ThreadPoolExecutor(max_workers=num_workers) as executor:
# Submit chunk of tasks
future_to_page = {
executor.submit(
process_page, temp_dir, pdf_path, page_info
): page_info
for page_info in chunk
}
# Process completed tasks as they finish
for future in as_completed(future_to_page):
try:
result = future.result(timeout=60) # Increased timeout to 60 seconds per page
if result.get('success', False):
processed_chunk.append(result)
page_info = future_to_page[future]
logger.debug(f"Processed page {page_info[1] + 1}", {
"success": result.get('success', False),
"processing_time": time.time() - start_time
})
except TimeoutError:
page_info = future_to_page[future]
logger.error(f"Timeout processing page {page_info[1] + 1}")
except Exception as e:
page_info = future_to_page[future]
logger.error(f"Error processing page {page_info[1] + 1}: {str(e)}")
chunk_time = time.time() - start_time
logger.info(f"Completed chunk {current_chunk_num}/{total_chunks}", {
"processed_pages": len(processed_chunk),
"chunk_processing_time": chunk_time,
"avg_time_per_page": chunk_time / len(chunk) if chunk else 0
})
all_processed_pages.extend(processed_chunk)
# Small delay between chunks to allow other tasks to process
await asyncio.sleep(0.1)
return all_processed_pages
@router.post("/convert")
async def convert_pdf_to_images(file: UploadFile = File(...)):
try:
async with processing_semaphore: # Control concurrent processing
start_time = time.time()
# Log request details
logger.info(
"Received file upload request",
{
"filename": file.filename,
"content_type": file.content_type,
"current_memory_usage_gb": psutil.Process()
.memory_info()
.rss
/ (1024**3),
"cpu_percent": psutil.cpu_percent(interval=1),
},
)
# Validate file
if not file.filename.endswith('.pdf'):
logger.error("Invalid file type")
return JSONResponse({
"status": "error",
"message": "Invalid file type. Please upload a .pdf file"
}, status_code=400)
# Create a temporary directory to store the PDF file
with tempfile.TemporaryDirectory() as temp_dir:
pdf_path = Path(temp_dir) / "document.pdf"
logger.debug(f"Saving file to temporary path: {pdf_path}")
try:
# Save uploaded file
content = await file.read()
logger.debug(f"Read file content, size: {len(content)} bytes")
with open(pdf_path, "wb") as buffer:
buffer.write(content)
logger.debug("File saved successfully")
if not pdf_path.exists() or pdf_path.stat().st_size == 0:
raise Exception("Failed to save file or file is empty")
# Get number of pages using pdfinfo
result = subprocess.run(['pdfinfo', str(pdf_path)], capture_output=True, text=True)
pages_line = [line for line in result.stdout.split('\n') if line.startswith('Pages:')][0]
num_pages = int(pages_line.split(':')[1].strip())
visible_pages = [(i, i) for i in range(num_pages)]
if num_pages == 0:
logger.warning("No pages found in document")
return JSONResponse({
"status": "error",
"message": "No pages found in document"
}, status_code=400)
logger.info(f"Processing {num_pages} pages")
# Calculate chunk size based on number of pages
chunk_size = min(5, max(2, math.ceil(num_pages / 4)))
processed_pages = await process_pages_in_chunks(str(temp_dir), str(pdf_path), visible_pages, chunk_size)
if not processed_pages:
raise Exception("Failed to process any pages successfully")
# Sort pages by index
processed_pages.sort(key=lambda x: x['index'])
logger.info(f"Successfully processed {len(processed_pages)} pages")
# After processing all pages
total_time = time.time() - start_time
logger.info("PDF processing completed", {
"total_processing_time": total_time,
"pages_processed": len(processed_pages),
"avg_time_per_page": total_time / len(processed_pages) if processed_pages else 0,
"final_memory_usage_gb": psutil.Process().memory_info().rss / (1024**3)
})
return JSONResponse({
"status": "success",
"slides": processed_pages, # Using same format as PowerPoint for consistency
"processing_stats": {
"total_time": total_time,
"pages_processed": len(processed_pages),
"avg_time_per_page": total_time / len(processed_pages) if processed_pages else 0
}
})
except Exception as inner_error:
logger.error(f"Inner error: {str(inner_error)}")
logger.error(traceback.format_exc())
raise
except Exception as e:
logger.error(f"Error processing PDF: {str(e)}")
logger.error(f"Python version: {sys.version}")
logger.error(f"Traceback: {traceback.format_exc()}")
return JSONResponse({
"status": "error",
"message": f"Failed to process PDF: {str(e)}"
}, status_code=500)
+398
View File
@@ -0,0 +1,398 @@
import os
from modules.logger_tool import initialise_logger
logger = initialise_logger(log_name="powerpoint", log_level=os.getenv("LOG_LEVEL"), log_dir=os.getenv("LOG_PATH"), log_format="default", runtime=True)
from fastapi import APIRouter, UploadFile, File, HTTPException
from fastapi.responses import JSONResponse
from pathlib import Path
import tempfile
from pptx import Presentation
from PIL import Image
import io
import base64
import traceback
import sys
import subprocess
from concurrent.futures import ThreadPoolExecutor, as_completed, TimeoutError
import asyncio
import psutil
import math
import time
router = APIRouter()
# Global semaphore to control total concurrent PowerPoint processing
MAX_CONCURRENT_PROCESSING = 4 # Adjust based on server capacity
processing_semaphore = asyncio.Semaphore(MAX_CONCURRENT_PROCESSING)
def calculate_optimal_workers():
"""Calculate optimal number of worker threads based on system resources."""
cpu_count = os.cpu_count() or 4
available_memory = psutil.virtual_memory().available
memory_per_worker = 500 * 1024 * 1024 # 500MB per worker estimate
# Calculate workers based on CPU and memory constraints
cpu_based_workers = max(1, cpu_count - 1) # Leave one core free
memory_based_workers = max(1, int(available_memory / memory_per_worker))
# Take the minimum of CPU and memory-based calculations
optimal_workers = min(cpu_based_workers, memory_based_workers)
# Cap at a reasonable maximum
final_workers = min(optimal_workers, 8) # Maximum 8 workers per process
# Log resource information
logger.info("Resource utilization:", {
"total_cpus": cpu_count,
"available_memory_gb": available_memory / (1024**3),
"cpu_based_workers": cpu_based_workers,
"memory_based_workers": memory_based_workers,
"final_workers": final_workers
})
return final_workers
def extract_text_from_shape(shape):
"""Extract text from a PowerPoint shape."""
if hasattr(shape, 'text') and shape.text.strip():
return shape.text.strip()
# Handle tables
if shape.has_table:
table_text = []
for row in shape.table.rows:
row_text = []
row_text.extend(cell.text.strip() for cell in row.cells if cell.text.strip())
if row_text:
table_text.append('| ' + ' | '.join(row_text) + ' |')
if table_text:
# Add markdown table header separator
table_text.insert(1, '|' + '---|' * (len(table_text[0].split('|')) - 2))
return '\n'.join(table_text)
# Handle grouped shapes
if hasattr(shape, 'shapes'):
group_text = []
for subshape in shape.shapes:
if text := extract_text_from_shape(subshape):
group_text.append(text)
return '\n'.join(group_text) if group_text else ''
return ''
def extract_slide_text(slide):
"""Extract text from a PowerPoint slide and format as markdown."""
slide_text = []
# Extract title if present
if slide.shapes.title and slide.shapes.title.text.strip():
slide_text.append(f"# {slide.shapes.title.text.strip()}")
# Process all shapes
for shape in slide.shapes:
if shape != slide.shapes.title: # Skip title as we've already processed it
if text := extract_text_from_shape(shape):
slide_text.append(text)
return '\n\n'.join(slide_text)
def process_slide(temp_dir: str, pdf_path: str, pptx_path: str, slide_info: tuple, timeout: int = 30) -> dict:
"""
Worker function to process a single slide and enforce 16:9 aspect ratio.
Args:
temp_dir: Path to temporary directory
pdf_path: Path to PDF file
pptx_path: Path to PowerPoint file
slide_info: Tuple of (index, slide_number)
timeout: Maximum time in seconds to process a single slide
Returns:
dict: Processed slide information
"""
i, slide_idx = slide_info
slide_num = slide_idx + 1 # PDF pages are 1-indexed
output_prefix = str(Path(temp_dir) / f"slide_{slide_num}")
try:
# Extract text from PowerPoint slide
prs = Presentation(pptx_path)
slide_text = extract_slide_text(prs.slides[slide_idx])
# Convert PDF page to PNG with timeout
process = subprocess.Popen(
[
'pdftoppm',
'-png',
'-singlefile',
'-f',
str(slide_num),
'-l',
str(slide_num),
'-r',
'600', # High resolution for better quality
pdf_path,
output_prefix,
],
stdout=subprocess.PIPE,
stderr=subprocess.PIPE
)
try:
stdout, stderr = process.communicate(timeout=timeout)
except subprocess.TimeoutExpired:
process.kill()
raise TimeoutError(f"Slide {slide_num} processing timed out after {timeout} seconds")
if process.returncode != 0:
raise Exception(f"pdftoppm failed for slide {slide_num}: {stderr.decode()}")
output_file = f"{output_prefix}.png"
if not Path(output_file).exists():
raise Exception(f"Could not find output file for slide {slide_num}")
# Open and process the image
with Image.open(output_file) as img:
result = _process_image(img, i)
if result['success']:
result['meta'] = {
'text': slide_text,
'format': 'markdown'
}
return result
except Exception as e:
logger.error(f"Error processing slide {slide_num}: {str(e)}")
return {
"index": i,
"error": str(e),
"success": False,
}
def _process_image(img: Image.Image, index: int) -> dict:
"""Process a single image, enforcing aspect ratio and size constraints."""
try:
# Enforce 16:9 aspect ratio
target_aspect_ratio = 16 / 9
img_aspect_ratio = img.width / img.height
if img_aspect_ratio > target_aspect_ratio: # Wider than 16:9
new_width = int(img.height * target_aspect_ratio)
offset = (img.width - new_width) // 2
img = img.crop((offset, 0, offset + new_width, img.height))
elif img_aspect_ratio < target_aspect_ratio: # Taller than 16:9
new_height = int(img.width / target_aspect_ratio)
offset = (img.height - new_height) // 2
img = img.crop((0, offset, img.width, offset + new_height))
# Resize to target resolution (2560x1440)
img = img.resize((2560, 1440), Image.Resampling.LANCZOS)
# Convert to base64
buffered = io.BytesIO()
img.save(buffered, format="PNG", optimize=True)
img_str = base64.b64encode(buffered.getvalue()).decode()
return {
"index": index,
"data": f"data:image/png;base64,{img_str}",
"success": True,
}
except Exception as e:
logger.error(f"Error processing image for slide {index}: {str(e)}")
return {
"index": index,
"error": str(e),
"success": False,
}
async def process_slides_in_chunks(temp_dir: str, pdf_path: str, pptx_path: str, visible_slides: list, chunk_size: int = 5):
"""Process slides in chunks to manage memory better."""
all_processed_slides = []
num_workers = calculate_optimal_workers()
total_chunks = math.ceil(len(visible_slides) / chunk_size)
logger.info("Starting slide processing:", {
"total_slides": len(visible_slides),
"chunk_size": chunk_size,
"total_chunks": total_chunks,
"workers_per_chunk": num_workers
})
# Process slides in chunks
for chunk_index in range(0, len(visible_slides), chunk_size):
chunk = visible_slides[chunk_index:chunk_index + chunk_size]
processed_chunk = []
current_chunk_num = (chunk_index // chunk_size) + 1
logger.info(f"Processing chunk {current_chunk_num}/{total_chunks}", {
"chunk_size": len(chunk),
"chunk_start_index": chunk_index,
"memory_usage_gb": psutil.Process().memory_info().rss / (1024**3)
})
start_time = time.time()
with ThreadPoolExecutor(max_workers=num_workers) as executor:
# Submit chunk of tasks
future_to_slide = {
executor.submit(
process_slide, temp_dir, pdf_path, pptx_path, slide_info
): slide_info
for slide_info in chunk
}
# Process completed tasks as they finish
for future in as_completed(future_to_slide):
try:
result = future.result(timeout=60) # Increased timeout to 60 seconds per slide
if result.get('success', False):
processed_chunk.append(result)
slide_info = future_to_slide[future]
logger.debug(f"Processed slide {slide_info[1] + 1}", {
"success": result.get('success', False),
"processing_time": time.time() - start_time
})
except TimeoutError:
slide_info = future_to_slide[future]
logger.error(f"Timeout processing slide {slide_info[1] + 1}")
except Exception as e:
slide_info = future_to_slide[future]
logger.error(f"Error processing slide {slide_info[1] + 1}: {str(e)}")
chunk_time = time.time() - start_time
logger.info(f"Completed chunk {current_chunk_num}/{total_chunks}", {
"processed_slides": len(processed_chunk),
"chunk_processing_time": chunk_time,
"avg_time_per_slide": chunk_time / len(chunk) if chunk else 0
})
all_processed_slides.extend(processed_chunk)
# Small delay between chunks to allow other tasks to process
await asyncio.sleep(0.1)
return all_processed_slides
@router.post("/convert")
async def convert_pptx_to_images(file: UploadFile = File(...)):
try:
async with processing_semaphore: # Control concurrent processing
start_time = time.time()
# Log request details
logger.info(
"Received file upload request",
{
"filename": file.filename,
"content_type": file.content_type,
"current_memory_usage_gb": psutil.Process()
.memory_info()
.rss
/ (1024**3),
"cpu_percent": psutil.cpu_percent(interval=1),
},
)
# Validate file
if not file.filename.endswith('.pptx'):
logger.error("Invalid file type")
return JSONResponse({
"status": "error",
"message": "Invalid file type. Please upload a .pptx file"
}, status_code=400)
# Create a temporary directory to store the PowerPoint file
with tempfile.TemporaryDirectory() as temp_dir:
pptx_path = Path(temp_dir) / "presentation.pptx"
logger.debug(f"Saving file to temporary path: {pptx_path}")
try:
# Save uploaded file
content = await file.read()
logger.debug(f"Read file content, size: {len(content)} bytes")
with open(pptx_path, "wb") as buffer:
buffer.write(content)
logger.debug("File saved successfully")
if not pptx_path.exists() or pptx_path.stat().st_size == 0:
raise Exception("Failed to save file or file is empty")
# Open the presentation and get visible slides
prs = Presentation(str(pptx_path))
visible_slides = [
(i, slide_idx)
for i, (slide_idx, _) in enumerate(
(i, slide)
for i, slide in enumerate(prs.slides)
if not hasattr(slide, 'show') or slide.show
)
]
num_slides = len(visible_slides)
if num_slides == 0:
logger.warning("No visible slides found in presentation")
return JSONResponse({
"status": "error",
"message": "No visible slides found in presentation"
}, status_code=400)
logger.info(f"Processing {num_slides} visible slides")
# Convert PowerPoint to PDF
pdf_path = Path(temp_dir) / "presentation.pdf"
logger.debug("Converting PowerPoint to PDF")
result = subprocess.run([
'soffice',
'--headless',
'--convert-to', 'pdf',
'--outdir', str(temp_dir),
str(pptx_path)
], check=True, capture_output=True, text=True)
if not pdf_path.exists():
raise Exception("PDF file was not created")
logger.debug(f"PDF created successfully at {pdf_path}, size: {pdf_path.stat().st_size} bytes")
# Calculate chunk size based on number of slides
chunk_size = min(5, max(2, math.ceil(num_slides / 4)))
processed_slides = await process_slides_in_chunks(str(temp_dir), str(pdf_path), str(pptx_path), visible_slides, chunk_size)
if not processed_slides:
raise Exception("Failed to process any slides successfully")
# Sort slides by index
processed_slides.sort(key=lambda x: x['index'])
logger.info(f"Successfully processed {len(processed_slides)} slides")
# After processing all slides
total_time = time.time() - start_time
logger.info("PowerPoint processing completed", {
"total_processing_time": total_time,
"slides_processed": len(processed_slides),
"avg_time_per_slide": total_time / len(processed_slides) if processed_slides else 0,
"final_memory_usage_gb": psutil.Process().memory_info().rss / (1024**3)
})
return JSONResponse({
"status": "success",
"slides": processed_slides,
"processing_stats": {
"total_time": total_time,
"slides_processed": len(processed_slides),
"avg_time_per_slide": total_time / len(processed_slides) if processed_slides else 0
}
})
except Exception as inner_error:
logger.error(f"Inner error: {str(inner_error)}")
logger.error(traceback.format_exc())
raise
except Exception as e:
logger.error(f"Error processing PowerPoint: {str(e)}")
logger.error(f"Python version: {sys.version}")
logger.error(f"Traceback: {traceback.format_exc()}")
return JSONResponse({
"status": "error",
"message": f"Failed to process PowerPoint: {str(e)}"
}, status_code=500)
View File
+418
View File
@@ -0,0 +1,418 @@
import os
from modules.logger_tool import initialise_logger
logger = initialise_logger(log_name="word", log_level=os.getenv("LOG_LEVEL"), log_dir=os.getenv("LOG_PATH"), log_format="default", runtime=True)
from fastapi import APIRouter, UploadFile, File, HTTPException
from fastapi.responses import JSONResponse
from pathlib import Path
import tempfile
from PIL import Image
import io
import base64
import traceback
import sys
import subprocess
from concurrent.futures import ThreadPoolExecutor, as_completed, TimeoutError
import asyncio
import psutil
import math
import time
from docx import Document
router = APIRouter()
# Global semaphore to control total concurrent Word processing
MAX_CONCURRENT_PROCESSING = 4 # Adjust based on server capacity
processing_semaphore = asyncio.Semaphore(MAX_CONCURRENT_PROCESSING)
def calculate_optimal_workers():
"""Calculate optimal number of worker threads based on system resources."""
cpu_count = os.cpu_count() or 4
available_memory = psutil.virtual_memory().available
memory_per_worker = 500 * 1024 * 1024 # 500MB per worker estimate
# Calculate workers based on CPU and memory constraints
cpu_based_workers = max(1, cpu_count - 1) # Leave one core free
memory_based_workers = max(1, int(available_memory / memory_per_worker))
# Take the minimum of CPU and memory-based calculations
optimal_workers = min(cpu_based_workers, memory_based_workers)
# Cap at a reasonable maximum
final_workers = min(optimal_workers, 8) # Maximum 8 workers per process
logger.info("Resource utilization:", {
"total_cpus": cpu_count,
"available_memory_gb": available_memory / (1024**3),
"cpu_based_workers": cpu_based_workers,
"memory_based_workers": memory_based_workers,
"final_workers": final_workers
})
return final_workers
def extract_text_from_paragraph(paragraph):
"""Extract text from a Word paragraph and format as markdown."""
text = paragraph.text.strip()
if not text:
return ''
# Handle different heading levels
if paragraph.style.name.startswith('Heading'):
level = int(paragraph.style.name[-1])
return f"{'#' * level} {text}"
# Handle lists
if paragraph._element.pPr is not None and paragraph._element.pPr.numPr is not None:
return f"* {text}"
return text
def extract_text_from_table(table):
"""Extract text from a Word table and format as markdown."""
# Process header row
header_row = []
header_row.extend((cell.text.strip() or ' ') for cell in table.rows[0].cells)
table_text = [
'| ' + ' | '.join(header_row) + ' |',
'|' + '---|' * (len(header_row) - 1) + '---|',
]
# Process remaining rows
for row in table.rows[1:]:
row_text = []
row_text.extend((cell.text.strip() or ' ') for cell in row.cells)
table_text.append('| ' + ' | '.join(row_text) + ' |')
return '\n'.join(table_text)
def extract_page_text(doc, page_index):
"""Extract text from a Word document page and format as markdown."""
# Note: python-docx doesn't provide direct page access, so we'll use a heuristic
# to group paragraphs into pages based on content length
CHARS_PER_PAGE = 3000 # Approximate characters per page
all_blocks = []
current_chars = 0
current_page = 0
for element in doc.element.body:
if current_page > page_index:
break
if element.tag.endswith('p'):
paragraph = doc.paragraphs[len(all_blocks)]
if text := extract_text_from_paragraph(paragraph):
current_chars += len(text)
if current_page == page_index:
all_blocks.append(text)
elif element.tag.endswith('tbl'):
table = doc.tables[sum(isinstance(b, str) for b in all_blocks)]
if text := extract_text_from_table(table):
current_chars += len(text)
if current_page == page_index:
all_blocks.append(text)
if current_chars >= CHARS_PER_PAGE:
current_page += 1
current_chars = 0
return '\n\n'.join(all_blocks)
def process_page(temp_dir: str, pdf_path: str, docx_path: str, page_info: tuple, timeout: int = 30) -> dict:
"""
Worker function to process a single page and maintain A4 proportions.
Args:
temp_dir: Path to temporary directory
pdf_path: Path to PDF file
docx_path: Path to Word file
page_info: Tuple of (index, page_number)
timeout: Maximum time in seconds to process a single page
Returns:
dict: Processed page information
"""
i, page_idx = page_info
page_num = page_idx + 1 # PDF pages are 1-indexed
output_prefix = str(Path(temp_dir) / f"page_{page_num}")
try:
# Extract text from Word document
doc = Document(docx_path)
page_text = extract_page_text(doc, page_idx)
# Convert PDF page to PNG with timeout
process = subprocess.Popen(
[
'pdftoppm',
'-png',
'-singlefile',
'-f',
str(page_num),
'-l',
str(page_num),
'-r',
'600', # High resolution for better quality
pdf_path,
output_prefix,
],
stdout=subprocess.PIPE,
stderr=subprocess.PIPE
)
try:
stdout, stderr = process.communicate(timeout=timeout)
except subprocess.TimeoutExpired:
process.kill()
raise TimeoutError(f"Page {page_num} processing timed out after {timeout} seconds")
if process.returncode != 0:
raise Exception(f"pdftoppm failed for page {page_num}: {stderr.decode()}")
output_file = f"{output_prefix}.png"
if not Path(output_file).exists():
raise Exception(f"Could not find output file for page {page_num}")
# Open and process the image
with Image.open(output_file) as img:
result = _process_image(img, i)
if result['success']:
result['meta'] = {
'text': page_text,
'format': 'markdown'
}
return result
except Exception as e:
logger.error(f"Error processing page {page_num}: {str(e)}")
return {
"index": i,
"error": str(e),
"success": False,
}
def _process_image(img: Image.Image, index: int) -> dict:
"""Process a single image, maintaining A4 proportions."""
try:
# Determine orientation and target dimensions
is_portrait = img.height > img.width
target_height = 720 # Fixed height to match frontend slide height
if is_portrait:
# A4 portrait ratio is 210:297
target_width = int(target_height * (210/297))
else:
# A4 landscape ratio is 297:210
target_width = int(target_height * (297/210))
# Resize image maintaining aspect ratio
img = img.resize((target_width, target_height), Image.Resampling.LANCZOS)
# Convert to base64
buffered = io.BytesIO()
img.save(buffered, format="PNG", optimize=True)
img_str = base64.b64encode(buffered.getvalue()).decode()
return {
"index": index,
"data": f"data:image/png;base64,{img_str}",
"success": True,
"dimensions": {
"width": target_width,
"height": target_height,
"orientation": "portrait" if is_portrait else "landscape"
}
}
except Exception as e:
logger.error(f"Error processing image for page {index}: {str(e)}")
return {
"index": index,
"error": str(e),
"success": False,
}
async def process_pages_in_chunks(temp_dir: str, pdf_path: str, docx_path: str, visible_pages: list, chunk_size: int = 5):
"""Process pages in chunks to manage memory better."""
all_processed_pages = []
num_workers = calculate_optimal_workers()
total_chunks = math.ceil(len(visible_pages) / chunk_size)
logger.info("Starting page processing:", {
"total_pages": len(visible_pages),
"chunk_size": chunk_size,
"total_chunks": total_chunks,
"workers_per_chunk": num_workers
})
# Process pages in chunks
for chunk_index in range(0, len(visible_pages), chunk_size):
chunk = visible_pages[chunk_index:chunk_index + chunk_size]
processed_chunk = []
current_chunk_num = (chunk_index // chunk_size) + 1
logger.info(f"Processing chunk {current_chunk_num}/{total_chunks}", {
"chunk_size": len(chunk),
"chunk_start_index": chunk_index,
"memory_usage_gb": psutil.Process().memory_info().rss / (1024**3)
})
start_time = time.time()
with ThreadPoolExecutor(max_workers=num_workers) as executor:
# Submit chunk of tasks
future_to_page = {
executor.submit(
process_page, temp_dir, pdf_path, docx_path, page_info
): page_info
for page_info in chunk
}
# Process completed tasks as they finish
for future in as_completed(future_to_page):
try:
result = future.result(timeout=60) # Increased timeout to 60 seconds per page
if result.get('success', False):
processed_chunk.append(result)
page_info = future_to_page[future]
logger.debug(f"Processed page {page_info[1] + 1}", {
"success": result.get('success', False),
"processing_time": time.time() - start_time
})
except TimeoutError:
page_info = future_to_page[future]
logger.error(f"Timeout processing page {page_info[1] + 1}")
except Exception as e:
page_info = future_to_page[future]
logger.error(f"Error processing page {page_info[1] + 1}: {str(e)}")
chunk_time = time.time() - start_time
logger.info(f"Completed chunk {current_chunk_num}/{total_chunks}", {
"processed_pages": len(processed_chunk),
"chunk_processing_time": chunk_time,
"avg_time_per_page": chunk_time / len(chunk) if chunk else 0
})
all_processed_pages.extend(processed_chunk)
# Small delay between chunks to allow other tasks to process
await asyncio.sleep(0.1)
return all_processed_pages
@router.post("/convert")
async def convert_docx_to_images(file: UploadFile = File(...)):
try:
async with processing_semaphore: # Control concurrent processing
start_time = time.time()
# Log request details
logger.info(
"Received file upload request",
{
"filename": file.filename,
"content_type": file.content_type,
"current_memory_usage_gb": psutil.Process()
.memory_info()
.rss
/ (1024**3),
"cpu_percent": psutil.cpu_percent(interval=1),
},
)
# Validate file
if not file.filename.endswith('.docx'):
logger.error("Invalid file type")
return JSONResponse({
"status": "error",
"message": "Invalid file type. Please upload a .docx file"
}, status_code=400)
# Create a temporary directory to store the Word file
with tempfile.TemporaryDirectory() as temp_dir:
docx_path = Path(temp_dir) / "document.docx"
pdf_path = Path(temp_dir) / "document.pdf"
logger.debug(f"Saving file to temporary path: {docx_path}")
try:
# Save uploaded file
content = await file.read()
logger.debug(f"Read file content, size: {len(content)} bytes")
with open(docx_path, "wb") as buffer:
buffer.write(content)
logger.debug("File saved successfully")
if not docx_path.exists() or docx_path.stat().st_size == 0:
raise Exception("Failed to save file or file is empty")
# Convert Word to PDF using LibreOffice
logger.debug("Converting Word to PDF")
result = subprocess.run([
'soffice',
'--headless',
'--convert-to', 'pdf',
'--outdir', str(temp_dir),
str(docx_path)
], check=True, capture_output=True, text=True)
if not pdf_path.exists():
raise Exception("PDF file was not created")
logger.debug(f"PDF created successfully at {pdf_path}, size: {pdf_path.stat().st_size} bytes")
# Get number of pages using pdfinfo
result = subprocess.run(['pdfinfo', str(pdf_path)], capture_output=True, text=True)
pages_line = [line for line in result.stdout.split('\n') if line.startswith('Pages:')][0]
num_pages = int(pages_line.split(':')[1].strip())
visible_pages = [(i, i) for i in range(num_pages)]
if num_pages == 0:
logger.warning("No pages found in document")
return JSONResponse({
"status": "error",
"message": "No pages found in document"
}, status_code=400)
logger.info(f"Processing {num_pages} pages")
# Calculate chunk size based on number of pages
chunk_size = min(5, max(2, math.ceil(num_pages / 4)))
processed_pages = await process_pages_in_chunks(str(temp_dir), str(pdf_path), str(docx_path), visible_pages, chunk_size)
if not processed_pages:
raise Exception("Failed to process any pages successfully")
# Sort pages by index
processed_pages.sort(key=lambda x: x['index'])
logger.info(f"Successfully processed {len(processed_pages)} pages")
# After processing all pages
total_time = time.time() - start_time
logger.info("Word document processing completed", {
"total_processing_time": total_time,
"pages_processed": len(processed_pages),
"avg_time_per_page": total_time / len(processed_pages) if processed_pages else 0,
"final_memory_usage_gb": psutil.Process().memory_info().rss / (1024**3)
})
return JSONResponse({
"status": "success",
"slides": processed_pages, # Using same format as PowerPoint for consistency
"processing_stats": {
"total_time": total_time,
"pages_processed": len(processed_pages),
"avg_time_per_page": total_time / len(processed_pages) if processed_pages else 0
}
})
except Exception as inner_error:
logger.error(f"Inner error: {str(inner_error)}")
logger.error(traceback.format_exc())
raise
except Exception as e:
logger.error(f"Error processing Word document: {str(e)}")
logger.error(f"Python version: {sys.version}")
logger.error(f"Traceback: {traceback.format_exc()}")
return JSONResponse({
"status": "error",
"message": f"Failed to process Word document: {str(e)}"
}, status_code=500)