Skip to content
Merged
2 changes: 1 addition & 1 deletion ocr_service/main.py
Original file line number Diff line number Diff line change
Expand Up @@ -69,7 +69,7 @@ def start_processing(item: ProcessItem):
except ocrmypdf.UnsupportedImageFormatError:
logger.error(f"Unsupported image format: {item.local_file_path}")
item.ocr_status = OCRStatus.UNSUPPORTED
ocr_error = "Unsupported image format"
ocr_error = OCRStatus.UNSUPPORTED.value
except ocrmypdf.DpiError as dpiex:
logger.error(f"DPI error: {item.local_file_path} {dpiex}")
item.ocr_status = OCRStatus.DPI_ERROR
Expand Down
47 changes: 26 additions & 21 deletions scansynclib/scansynclib/ProcessItem.py
Original file line number Diff line number Diff line change
Expand Up @@ -59,29 +59,31 @@ def get_progress(cls, status: ProcessStatus) -> int:
class OCRStatus(Enum):
"""Enumeration of possible OCR statuses.

UNKNOWN: OCR status not yet determined.
PENDING: Item is queued for OCR.
PROCESSING: OCR is currently running on the item.
COMPLETED: OCR completed successfully on the item.
FAILED: OCR failed on the item.
SKIPPED: Item was skipped and OCR was not performed.
UNSUPPORTED: Item type is not supported for OCR.
DPI_ERROR: Image DPI is too low for accurate OCR.
INPUT_ERROR: Error reading input image/PDF.
OUTPUT_ERROR: Error writing OCR output file.
Each value is the human-readable description shown in the web UI as the status badge.

UNKNOWN: Status has not yet been determined for this item.
PENDING: Item is queued and waiting for OCR processing to begin.
PROCESSING: OCR is actively running on the item.
COMPLETED: OCR finished successfully and text was extracted.
FAILED: OCR finished but the output was unusable.
SKIPPED: OCR was not performed on this item.
UNSUPPORTED: The image format is not supported by the OCR engine.
DPI_ERROR: The image resolution is too low for accurate OCR.
INPUT_ERROR: The input file could not be read by the OCR engine.
OUTPUT_ERROR: The OCR engine could not write the output file.
NO_TEXT: OCR completed but the output file contained no extractable text.
"""
UNKNOWN = 0
PENDING = 1
PROCESSING = 2
COMPLETED = 3
FAILED = -1
SKIPPED = -2
UNSUPPORTED = -3
DPI_ERROR = -4
INPUT_ERROR = -5
OUTPUT_ERROR = -6
NO_TEXT = -7
UNKNOWN = "Unknown"
PENDING = "Waiting for OCR"
PROCESSING = "OCR in progress"
COMPLETED = "OCR completed"
FAILED = "OCR failed"
SKIPPED = "OCR skipped"
UNSUPPORTED = "Unsupported image format"
DPI_ERROR = "Image DPI too low for accurate OCR"
INPUT_ERROR = "Error reading input file"
OUTPUT_ERROR = "Error writing OCR output file"
NO_TEXT = "No text found in OCR output"


class FileNamingStatus(Enum):
Expand Down Expand Up @@ -185,6 +187,9 @@ def __init__(self, local_file_path: str, item_type: ItemType, status: ProcessSta
self.file_naming_db_id = None
"""The ID of the file naming entry in the database, if applicable."""

self.sync_db_id = None
"""The ID of the sync (upload) entry in the database, if applicable."""

self.file_naming_status = FileNamingStatus.PENDING
"""The status of the file naming process."""

Expand Down
10 changes: 10 additions & 0 deletions scansynclib/scansynclib/db/schema.sql
Original file line number Diff line number Diff line change
Expand Up @@ -42,4 +42,14 @@ CREATE TABLE IF NOT EXISTS file_naming_jobs (
file_naming_status TEXT NOT NULL,
success Boolean NOT NULL DEFAULT 0,
error_description TEXT
);

CREATE TABLE IF NOT EXISTS sync_jobs (
id INTEGER PRIMARY KEY AUTOINCREMENT,
scanneddata_id INTEGER NOT NULL,
started DATETIME NOT NULL DEFAULT (DATETIME('now', 'localtime')),
finished DATETIME,
sync_status TEXT NOT NULL,
success Boolean NOT NULL DEFAULT 0,
error_description TEXT
);
167 changes: 167 additions & 0 deletions tests/test_logs_api.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,167 @@
"""Tests for the OCR and sync logging API endpoints."""

import json
import pytest
import sys
import os
from unittest.mock import patch, MagicMock

# Add paths for imports
sys.path.insert(0, os.path.join(os.path.dirname(__file__), '../scansynclib'))
sys.path.insert(0, os.path.join(os.path.dirname(__file__), '../web_service/src'))

# Ensure the data directory exists for sqlite_wrapper module-level initialization
os.makedirs(os.path.join(os.path.dirname(__file__), '../data'), exist_ok=True)

# Mock Redis before any scansynclib imports, since settings.py connects at module level
import redis as _real_redis


def _mock_from_url(*args, **kwargs):
mock_client = MagicMock()
mock_client.get.return_value = None # No existing settings in Redis
mock_client.set.return_value = True
mock_client.publish.return_value = 0
mock_pubsub = MagicMock()
mock_pubsub.subscribe.return_value = None
mock_pubsub.listen.return_value = iter([]) # Empty iterator
mock_client.pubsub.return_value = mock_pubsub
return mock_client


@pytest.fixture(scope="session", autouse=True)
def mock_redis_from_url():
"""Patch redis.Redis.from_url for the entire test session and restore afterwards."""
orig = _real_redis.Redis.from_url
_real_redis.Redis.from_url = _mock_from_url
yield
_real_redis.Redis.from_url = orig


@pytest.fixture
def app():
"""Create a Flask test app with the api blueprint."""
from flask import Flask
from routes.api import api_bp

app = Flask(__name__)
app.register_blueprint(api_bp)
app.config['TESTING'] = True
return app


@pytest.fixture
def client(app):
"""Create a Flask test client."""
return app.test_client()


class TestOcrLogsAPI:
"""Test cases for the /api/ocr-logs endpoint."""

def test_ocr_logs_returns_paginated_data(self, client):
logs = [
{
'id': 2,
'scanneddata_id': 5,
'started': '2024-06-01 12:00:00',
'finished': '2024-06-01 12:01:00',
'ocr_status': 'COMPLETED',
'ocr_error': None,
'file_name': 'invoice.pdf',
}
]
with patch('routes.api.execute_query') as mock_query:
mock_query.side_effect = [1, logs] # count, logs
response = client.get('/api/ocr-logs?page=1&per_page=20')
data = json.loads(response.data)

assert response.status_code == 200
assert data['total_count'] == 1
assert data['total_pages'] == 1
assert data['page'] == 1
assert data['logs'][0]['ocr_status'] == 'COMPLETED'

def test_ocr_logs_success_filter(self, client):
with patch('routes.api.execute_query') as mock_query:
mock_query.side_effect = [0, []]
client.get('/api/ocr-logs?filter=success')

count_query = mock_query.call_args_list[0].args[0]
assert "ocr_status = 'COMPLETED'" in count_query

def test_ocr_logs_failed_filter(self, client):
with patch('routes.api.execute_query') as mock_query:
mock_query.side_effect = [0, []]
client.get('/api/ocr-logs?filter=failed')

count_query = mock_query.call_args_list[0].args[0]
assert "NOT IN ('COMPLETED', 'PROCESSING')" in count_query

def test_ocr_logs_handles_none_count(self, client):
with patch('routes.api.execute_query') as mock_query:
mock_query.side_effect = [None, []]
response = client.get('/api/ocr-logs')
data = json.loads(response.data)

assert response.status_code == 200
assert data['total_count'] == 0
assert data['total_pages'] == 0

def test_ocr_logs_database_error_returns_500(self, client):
with patch('routes.api.execute_query') as mock_query:
mock_query.side_effect = Exception("boom")
response = client.get('/api/ocr-logs')

assert response.status_code == 500


class TestSyncLogsAPI:
"""Test cases for the /api/sync-logs endpoint."""

def test_sync_logs_returns_paginated_data(self, client):
logs = [
{
'id': 7,
'scanneddata_id': 9,
'started': '2024-06-01 12:00:00',
'finished': '2024-06-01 12:02:00',
'sync_status': 'COMPLETED',
'success': 1,
'error_description': None,
'file_name': 'doc.pdf',
}
]
with patch('routes.api.execute_query') as mock_query:
mock_query.side_effect = [1, logs]
response = client.get('/api/sync-logs')
data = json.loads(response.data)

assert response.status_code == 200
assert data['total_count'] == 1
assert data['logs'][0]['sync_status'] == 'COMPLETED'

def test_sync_logs_success_filter(self, client):
with patch('routes.api.execute_query') as mock_query:
mock_query.side_effect = [0, []]
client.get('/api/sync-logs?filter=success')

count_query = mock_query.call_args_list[0].args[0]
assert "sync_jobs.success = 1" in count_query

def test_sync_logs_failed_filter(self, client):
with patch('routes.api.execute_query') as mock_query:
mock_query.side_effect = [0, []]
client.get('/api/sync-logs?filter=failed')

count_query = mock_query.call_args_list[0].args[0]
assert "sync_jobs.success = 0" in count_query

def test_sync_logs_pagination_math(self, client):
with patch('routes.api.execute_query') as mock_query:
mock_query.side_effect = [12, []]
response = client.get('/api/sync-logs?per_page=5')
data = json.loads(response.data)

assert data['total_count'] == 12
assert data['total_pages'] == 3
27 changes: 24 additions & 3 deletions upload_service/main.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,7 @@
from scansynclib.ProcessItem import ProcessItem, ProcessStatus
from scansynclib.logging import logger
from scansynclib.helpers import connect_rabbitmq, move_to_failed
from scansynclib.sqlite_wrapper import update_scanneddata_database
from scansynclib.sqlite_wrapper import update_scanneddata_database, execute_query
from scansynclib.onedrive_api import upload_small
from scansynclib.config import config
import os
Expand All @@ -14,25 +14,44 @@
RABBITQUEUE = "upload_queue"


def finalize_sync_job(item: ProcessItem, error: str = None):
"""Persist the final state of a sync (upload) job to the sync_jobs table."""
success = 1 if item.status == ProcessStatus.COMPLETED else 0
execute_query(
"UPDATE sync_jobs SET sync_status = ?, success = ?, error_description = ?, finished = DATETIME('now', 'localtime') WHERE id = ?",
(item.status.name, success, error, item.sync_db_id)
)


def callback(ch, method, properties, body):
item = None
try:
item: ProcessItem = pickle.loads(body)
if not isinstance(item, ProcessItem):
logger.warning("Received object, that is not of type ProcessItem. Skipping.")
return
logger.info(f"Received PDF for Upload: {item.filename}")
item.sync_db_id = execute_query(
"INSERT INTO sync_jobs (scanneddata_id, sync_status) VALUES (?, ?)",
(item.db_id, ProcessStatus.SYNC.name),
return_last_id=True
)
if not os.path.exists(item.ocr_file):
logger.error(f"OCR file does not exist for upload: {item.ocr_file}")
item.status = ProcessStatus.SYNC_FAILED
update_scanneddata_database(item, {"file_status": item.status.value})
finalize_sync_job(item, "OCR file does not exist for upload")
move_to_failed(item)
else:
start_processing(item)
ch.basic_ack(delivery_tag=method.delivery_tag)
except Exception:
logger.exception(f"Failed processing {body}.")
item.status = ProcessStatus.SYNC_FAILED
update_scanneddata_database(item, {"file_status": item.status.value})
if item is not None and isinstance(item, ProcessItem):
item.status = ProcessStatus.SYNC_FAILED
update_scanneddata_database(item, {"file_status": item.status.value})
if item.sync_db_id is not None:
finalize_sync_job(item, "Unexpected error during upload")


def start_processing(item: ProcessItem):
Expand Down Expand Up @@ -68,6 +87,7 @@ def start_processing(item: ProcessItem):
logger.error(f"Failed to upload {item.ocr_file}")
item.status = ProcessStatus.SYNC_FAILED
move_to_failed(item)
finalize_sync_job(item, "Failed to upload file to OneDrive")
else:
logger.info(f"Upload completed: {item.filename}")

Expand Down Expand Up @@ -95,6 +115,7 @@ def start_processing(item: ProcessItem):
logger.exception(f"Failed to delete additional local file {additional_path}")

item.status = ProcessStatus.COMPLETED
finalize_sync_job(item)
update_scanneddata_database(item, {"file_status": item.status.value})


Expand Down
Loading
Loading