diff --git a/.github/workflows/js-tests.yml b/.github/workflows/js-tests.yml new file mode 100644 index 0000000..f921e3a --- /dev/null +++ b/.github/workflows/js-tests.yml @@ -0,0 +1,31 @@ +name: Run JS Tests + +on: + push: + branches: + - main + paths: + - '**/*.js' + - '**/*.mjs' + pull_request: + branches: + - main + paths: + - '**/*.js' + - '**/*.mjs' + +jobs: + run-js-tests: + runs-on: ubuntu-latest + + steps: + - name: Checkout repository + uses: actions/checkout@v4 + + - name: Setup Node.js + uses: actions/setup-node@v4 + with: + node-version: 'latest' + + - name: Run JS tests + run: npm run test:js diff --git a/docker-compose.test.yml b/docker-compose.test.yml index 7890c87..9a214c2 100644 --- a/docker-compose.test.yml +++ b/docker-compose.test.yml @@ -41,6 +41,7 @@ services: - ./detection_service:/tests/tests/detection_service - ./web_service:/tests/tests/web_service - ./ocr_service:/tests/tests/ocr_service + - ./file_naming_service:/tests/tests/file_naming_service redis: image: redis:latest diff --git a/file_naming_service/main.py b/file_naming_service/main.py index ab8a19e..d8295cd 100644 --- a/file_naming_service/main.py +++ b/file_naming_service/main.py @@ -17,7 +17,22 @@ RABBITQUEUE = "file_naming_queue" +def get_latest_file_naming_status(item: ProcessItem): + status_name = execute_query( + "SELECT file_naming_status FROM file_naming_jobs WHERE id = ?", + (item.file_naming_db_id,), + return_scalar=True + ) + if status_name: + try: + return FileNamingStatus[status_name] + except KeyError: + logger.warning(f"Unknown file naming status '{status_name}' for item {item.filename}") + return item.file_naming_status + + def callback(ch, method, properties, body): + item = None try: item: ProcessItem = pickle.loads(body) @@ -42,6 +57,7 @@ def callback(ch, method, properties, body): if not openai_enabled and not ollama_enabled: logger.error("Neither OpenAI nor Ollama is enabled. Please enable one of them in the settings.") + item.file_naming_status = FileNamingStatus.PROCESSING execute_query('UPDATE file_naming_jobs SET file_naming_status = ? WHERE id = ?', (FileNamingStatus.PROCESSING.name, item.file_naming_db_id)) method_setting = settings.file_naming.method @@ -65,7 +81,12 @@ def callback(ch, method, properties, body): execute_query("UPDATE file_naming_jobs SET file_naming_status = ?, error_description = ?, finished = DATETIME('now', 'localtime') WHERE id = ?", (FileNamingStatus.FAILED.name, "OCR file does not exist", item.file_naming_db_id)) except TypeError as e: logger.error(f"Received object is not a ProcessItem: {e}. Skipping.") - execute_query("UPDATE file_naming_jobs SET file_naming_status = ?, error_description = ?, finished = DATETIME('now', 'localtime') WHERE id = ?", (FileNamingStatus.FAILED.name, str(e), item.file_naming_db_id)) + item_file_naming_db_id = getattr(item, "file_naming_db_id", None) + if item_file_naming_db_id: + execute_query( + "UPDATE file_naming_jobs SET file_naming_status = ?, error_description = ?, finished = DATETIME('now', 'localtime') WHERE id = ?", + (FileNamingStatus.FAILED.name, str(e), item_file_naming_db_id) + ) return except Exception as e: logger.exception(f"Failed processing {item.filename}.") @@ -77,7 +98,10 @@ def callback(ch, method, properties, body): logger.error("Connection lost while acknowledging message. Reconnecting...") connection, channel = connect_rabbitmq([RABBITQUEUE], heartbeat=120) channel.basic_ack(delivery_tag=method.delivery_tag) - if item: + if isinstance(item, ProcessItem): + item_file_naming_db_id = getattr(item, "file_naming_db_id", None) + if item_file_naming_db_id: + item.file_naming_status = get_latest_file_naming_status(item) item.status = ProcessStatus.SYNC_PENDING update_scanneddata_database(item, {"file_status": item.status.value}) forward_to_rabbitmq("upload_queue", item) @@ -101,5 +125,6 @@ def start_consuming_with_reconnect(): time.sleep(5) -# Start the consumer with reconnect logic -start_consuming_with_reconnect() +if __name__ == "__main__": + # Start the consumer with reconnect logic + start_consuming_with_reconnect() diff --git a/ocr_service/main.py b/ocr_service/main.py index a32bd06..5ca51eb 100644 --- a/ocr_service/main.py +++ b/ocr_service/main.py @@ -1,6 +1,6 @@ from scansynclib.logging import logger from scansynclib.ProcessItem import ProcessItem, ProcessStatus, OCRStatus -from scansynclib.sqlite_wrapper import update_scanneddata_database +from scansynclib.sqlite_wrapper import execute_query, update_scanneddata_database from scansynclib.helpers import connect_rabbitmq, forward_to_rabbitmq, extract_text import pickle import ocrmypdf @@ -30,53 +30,77 @@ def callback(ch, method, properties, body): def start_processing(item: ProcessItem): item.status = ProcessStatus.OCR + item.ocr_status = OCRStatus.PROCESSING + item.ocr_db_id = execute_query( + 'INSERT INTO ocr_jobs (scanneddata_id, ocr_status) VALUES (?, ?)', + (item.db_id, OCRStatus.PROCESSING.name), + return_last_id=True + ) update_scanneddata_database(item, {"file_status": item.status.value}) item.time_ocr_started = datetime.now() logger.info(f"Processing file with OCR: {item.filename}") + ocr_error = None + result = None try: result = ocrmypdf.ocr(item.local_file_path, item.ocr_file, output_type='pdfa', skip_text=True, rotate_pages=True, jpg_quality=80, png_quality=80, optimize=2, language=["eng", "deu"], tesseract_timeout=120) logger.debug(f"OCR exited with code {result}") - + if result != 0: logger.error(f"OCR exited with code {result}") item.ocr_status = OCRStatus.FAILED + ocr_error = f"OCR exited with code {result}" else: logger.info(f"OCR processing completed: {item.filename}") - + # Verify that the OCR file actually contains text if os.path.exists(item.ocr_file): - extracted_text = (extract_text(item.ocr_file, max_pages=2, max_chars=2048) or "").strip() + extracted_text = (extract_text(item.ocr_file, max_pages=5, max_chars=2048) or "").strip() if extracted_text: logger.info(f"OCR verification successful: extracted {len(extracted_text)} characters from {item.filename}") item.ocr_status = OCRStatus.COMPLETED else: logger.warning(f"OCR verification failed: no text found in OCR output file {item.ocr_file}") - item.ocr_status = OCRStatus.FAILED + item.ocr_status = OCRStatus.NO_TEXT else: logger.error(f"OCR output file not found: {item.ocr_file}") item.ocr_status = OCRStatus.OUTPUT_ERROR except ocrmypdf.UnsupportedImageFormatError: logger.error(f"Unsupported image format: {item.local_file_path}") item.ocr_status = OCRStatus.UNSUPPORTED + ocr_error = "Unsupported image format" except ocrmypdf.DpiError as dpiex: logger.error(f"DPI error: {item.local_file_path} {dpiex}") item.ocr_status = OCRStatus.DPI_ERROR + ocr_error = str(dpiex) except ocrmypdf.InputFileError as inex: logger.error(f"Input error: {item.local_file_path} {inex}") item.ocr_status = OCRStatus.INPUT_ERROR + ocr_error = str(inex) except ocrmypdf.OutputFileAccessError as outex: logger.error(f"Output error: {item.local_file_path} {outex}") item.ocr_status = OCRStatus.OUTPUT_ERROR + ocr_error = str(outex) except ocrmypdf.MissingDependencyError: logger.exception("Cannot process with OCR due to missing dependencies.") item.ocr_status = OCRStatus.FAILED + ocr_error = "Missing OCR dependency" except Exception as ex: logger.exception(f"Failed processing {item.local_file_path} with OCR: {ex}") item.ocr_status = OCRStatus.FAILED + ocr_error = str(ex) finally: item.time_ocr_finished = datetime.now() + if result is not None and result != 0: + item.ocr_status = OCRStatus.FAILED + if not ocr_error: + ocr_error = f"OCR exited with code {result}" + if item.ocr_db_id: + execute_query( + "UPDATE ocr_jobs SET ocr_status = ?, ocr_error = ?, finished = DATETIME('now', 'localtime') WHERE id = ?", + (item.ocr_status.name, ocr_error, item.ocr_db_id) + ) item.status = ProcessStatus.SYNC_PENDING try: diff --git a/package.json b/package.json index 435236c..cf85011 100644 --- a/package.json +++ b/package.json @@ -8,7 +8,8 @@ "test": "tests" }, "scripts": { - "test": "echo \"Error: no test specified\" && exit 1" + "test": "echo \"Error: no test specified\" && exit 1", + "test:js": "node --test \"tests/js/**/*.test.mjs\"" }, "repository": { "type": "git", diff --git a/run-tests.sh b/run-tests.sh index b437ffb..8311d48 100755 --- a/run-tests.sh +++ b/run-tests.sh @@ -5,6 +5,9 @@ set -e COMPOSE_FILE="docker-compose.test.yml" TEST_SERVICE_NAME="test_service" +echo "๐Ÿงช Running JS tests..." +npm run test:js + echo "๐Ÿงช Starting tests with Docker Compose..." docker compose -f $COMPOSE_FILE up --build --abort-on-container-exit --exit-code-from $TEST_SERVICE_NAME diff --git a/scansynclib/scansynclib/ProcessItem.py b/scansynclib/scansynclib/ProcessItem.py index c64b76a..bca857a 100644 --- a/scansynclib/scansynclib/ProcessItem.py +++ b/scansynclib/scansynclib/ProcessItem.py @@ -69,6 +69,7 @@ class OCRStatus(Enum): DPI_ERROR: Image DPI is too low for accurate OCR. INPUT_ERROR: Error reading input image/PDF. OUTPUT_ERROR: Error writing OCR output file. + NO_TEXT: OCR completed but the output file contained no extractable text. """ UNKNOWN = 0 PENDING = 1 @@ -80,6 +81,7 @@ class OCRStatus(Enum): DPI_ERROR = -4 INPUT_ERROR = -5 OUTPUT_ERROR = -6 + NO_TEXT = -7 class FileNamingStatus(Enum): diff --git a/tests/js/dashboard.getStepStatuses.test.mjs b/tests/js/dashboard.getStepStatuses.test.mjs new file mode 100644 index 0000000..3e0088c --- /dev/null +++ b/tests/js/dashboard.getStepStatuses.test.mjs @@ -0,0 +1,179 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import { fileURLToPath } from "node:url"; +import { dirname, join } from "node:path"; +import vm from "node:vm"; + +// Load the real dashboard.js into an isolated context without modifying it. +// dashboard.js is a browser global script, so we stub the only top-level +// side effect (document.addEventListener) and read the pure functions back +// out of the sandbox global object. +const here = dirname(fileURLToPath(import.meta.url)); +const dashboardSource = readFileSync( + join(here, "..", "..", "web_service", "src", "static", "js", "dashboard.js"), + "utf8" +); + +const sandbox = { + document: { addEventListener() {} }, + console: { log() {}, warn() {}, error() {} } +}; +vm.createContext(sandbox); +vm.runInContext(dashboardSource, sandbox); + +const { getStepStatuses } = sandbox; + +test("getStepStatuses exists after loading dashboard.js", () => { + assert.equal(typeof getStepStatuses, "function"); +}); + +test("failed OCR is failed while file naming becomes the active (current) step", () => { + // file_status "File Name Pending" maps to progressStep 2, the same index as + // the OCR step. Processing continues after a non-fatal OCR failure, so the + // OCR step must be marked failed and the file naming step must become the + // active ("current") step so its marquee animation is shown. + const statuses = getStepStatuses(2, false, false, false, { + file_status: "File Name Pending", + ocr_status: "FAILED", + file_naming_status: "PENDING" + }); + + assert.deepEqual(Array.from(statuses), ["completed", "completed", "failed", "current", "pending"]); +}); + +test("sync pending marks upload as the active step (not file naming)", () => { + // "Sync Pending" maps to progressStep 3, but the upload step (index 4) is the + // one actually waiting to run. + const statuses = getStepStatuses(3, false, false, false, { + file_status: "Sync Pending", + ocr_status: "COMPLETED", + file_naming_status: "COMPLETED" + }); + + assert.deepEqual(Array.from(statuses), ["completed", "completed", "completed", "completed", "current"]); +}); + +test("failed OCR is shown as failed even when the document completed overall", () => { + const statuses = getStepStatuses(5, false, false, true, { + file_status: "Completed", + ocr_status: "FAILED" + }); + + assert.equal(statuses[2], "failed"); +}); + +test("completed doc with failed OCR and failed file naming (real reload data)", () => { + // Mirrors the real DB state after a non-fatal OCR failure: the document is + // uploaded (Completed) but ocr_jobs and file_naming_jobs both recorded a + // failure. Both stages must render red on page reload. + const statuses = getStepStatuses(5, false, false, true, { + file_status: "Completed", + status_progressbar: 5, + ocr_status: "FAILED", + file_naming_status: "RATE_LIMIT_ERROR" + }); + + assert.deepEqual(Array.from(statuses), ["completed", "completed", "failed", "failed", "completed"]); +}); + +test("non-failure OCR status keeps the OCR step as current when it is the active step", () => { + const statuses = getStepStatuses(2, false, false, false, { + file_status: "OCR Processing", + ocr_status: "PROCESSING" + }); + + assert.deepEqual(Array.from(statuses), ["completed", "completed", "current", "pending", "pending"]); +}); + +test("successful OCR shows the OCR step as completed once processing moved on", () => { + const statuses = getStepStatuses(3, false, false, false, { + file_status: "File Name Processing", + ocr_status: "COMPLETED", + file_naming_status: "PROCESSING" + }); + + assert.equal(statuses[2], "completed"); + assert.equal(statuses[3], "current"); +}); + +test("other OCR failure variants are surfaced as failed", () => { + for (const ocrStatus of ["NO_TEXT", "UNSUPPORTED", "DPI_ERROR", "INPUT_ERROR", "OUTPUT_ERROR"]) { + const statuses = getStepStatuses(2, false, false, false, { + file_status: "File Name Pending", + ocr_status: ocrStatus + }); + assert.equal(statuses[2], "failed", `expected failed for ${ocrStatus}`); + } +}); + +test("file naming failure is surfaced as failed at its step", () => { + const statuses = getStepStatuses(3, false, false, false, { + file_status: "Sync Pending", + ocr_status: "COMPLETED", + file_naming_status: "NO_OCR_FILE" + }); + + assert.equal(statuses[3], "failed"); +}); + +// Mirror how addPdfCard / updateProgressBar derive the branch flags from the data. +function deriveFlags(fileStatus, progressStep) { + const s = (fileStatus || "").toLowerCase(); + const isDeleted = s.includes("deleted"); + const isFailed = s.includes("failed") || isDeleted || progressStep === -1; + const isCompleted = s.includes("completed"); + return { isFailed, isDeleted, isCompleted }; +} + +function statusesFor(fileStatus, progressStep, extra = {}) { + const { isFailed, isDeleted, isCompleted } = deriveFlags(fileStatus, progressStep); + return getStepStatuses(progressStep, isFailed, isDeleted, isCompleted, { + file_status: fileStatus, + ...extra + }); +} + +// Every in-progress pipeline state must place exactly one "current" (marquee) +// segment on the stage the pipeline has actually reached. status_progressbar +// values mirror StatusProgressBar._progress_map on the server. +const inProgressCases = [ + { file_status: "File Not Ready", pb: 0, current: 0 }, + { file_status: "Reading Metadata", pb: 1, current: 1 }, + { file_status: "OCR Pending", pb: 1, current: 2 }, + { file_status: "OCR Processing", pb: 2, current: 2 }, + { file_status: "File Name Pending", pb: 2, current: 3 }, + { file_status: "File Name Processing", pb: 3, current: 3 }, + { file_status: "Sync Pending", pb: 3, current: 4 }, + { file_status: "Syncing", pb: 4, current: 4 } +]; + +for (const { file_status, pb, current } of inProgressCases) { + test(`in-progress "${file_status}" marks exactly one current step at index ${current}`, () => { + const statuses = statusesFor(file_status, pb); + const currentIndexes = Array.from(statuses).map((s, i) => (s === "current" ? i : -1)).filter((i) => i >= 0); + assert.deepEqual(currentIndexes, [current], `single marquee step expected at ${current}`); + // Everything before the current step is done, everything after is pending. + for (let i = 0; i < 5; i++) { + if (i < current) assert.equal(statuses[i], "completed", `step ${i} should be completed`); + if (i > current) assert.equal(statuses[i], "pending", `step ${i} should be pending`); + } + }); +} + +// Terminal states never show a marquee (no "current" segment). +const terminalCases = [ + { file_status: "Completed", pb: 5, expected: ["completed", "completed", "completed", "completed", "completed"] }, + { file_status: "Invalid File", pb: -1, expected: ["failed", "pending", "pending", "pending", "pending"] }, + { file_status: "Sync Failed", pb: -1, expected: ["completed", "completed", "completed", "completed", "failed"] }, + { file_status: "Deleted", pb: -1, expected: ["failed", "failed", "failed", "failed", "failed"] } +]; + +for (const { file_status, pb, expected } of terminalCases) { + test(`terminal "${file_status}" shows no marquee and renders as expected`, () => { + const statuses = statusesFor(file_status, pb, { ocr_status: "COMPLETED", file_naming_status: "COMPLETED" }); + assert.ok(!statuses.includes("current"), "terminal state must not show a current/marquee step"); + assert.deepEqual(Array.from(statuses), expected); + }); +} + diff --git a/tests/test_badge_generator.py b/tests/test_badge_generator.py index a54f679..b6aecc0 100644 --- a/tests/test_badge_generator.py +++ b/tests/test_badge_generator.py @@ -162,7 +162,7 @@ def test_none_local_filepath(self): except ImportError: # For Docker environment from badge_generator import _deterministic_hash - + expected_hash = _deterministic_hash('N/A') % len(SMB_TAG_COLORS) expected_color = SMB_TAG_COLORS[expected_hash] assert badges[0]['color'] == expected_color diff --git a/tests/test_file_naming_status.py b/tests/test_file_naming_status.py new file mode 100644 index 0000000..f177116 --- /dev/null +++ b/tests/test_file_naming_status.py @@ -0,0 +1,84 @@ +import pickle +import sys +import types + +import pytest + + +# Importing the service pulls in scansynclib.sqlite_wrapper, which initializes a +# real SQLite database at import time. That database is not available in the unit +# test environment, so replace the module with a stub. The query helpers are +# mocked per-test anyway. +_sqlite_stub = types.ModuleType("scansynclib.sqlite_wrapper") +_sqlite_stub.execute_query = lambda *args, **kwargs: None +_sqlite_stub.update_scanneddata_database = lambda *args, **kwargs: None + +_original_sqlite_wrapper = sys.modules.get("scansynclib.sqlite_wrapper") +sys.modules["scansynclib.sqlite_wrapper"] = _sqlite_stub + + +def teardown_module(module): + if _original_sqlite_wrapper is None: + sys.modules.pop("scansynclib.sqlite_wrapper", None) + else: + sys.modules["scansynclib.sqlite_wrapper"] = _original_sqlite_wrapper + + +import file_naming_service.main as fn_main # noqa: E402 +from scansynclib.ProcessItem import ProcessItem, ItemType, FileNamingStatus # noqa: E402 + + +@pytest.fixture +def item(tmp_path): + file_path = tmp_path / "doc.pdf" + file_path.write_bytes(b"%PDF-1.4 test") + process_item = ProcessItem(str(file_path), ItemType.PDF) + process_item.db_id = 7 + process_item.file_naming_db_id = 11 + process_item.file_naming_status = FileNamingStatus.PROCESSING + return process_item + + +@pytest.mark.parametrize( + "stored_status, expected", + [ + ("COMPLETED", FileNamingStatus.COMPLETED), + ("NO_SERVER_CONNECTION", FileNamingStatus.NO_SERVER_CONNECTION), + ("AUTHENTICATION_ERROR", FileNamingStatus.AUTHENTICATION_ERROR), + ], +) +def test_get_latest_status_returns_db_value(item, mocker, stored_status, expected): + mocker.patch.object(fn_main, "execute_query", return_value=stored_status) + + assert fn_main.get_latest_file_naming_status(item) == expected + + +def test_get_latest_status_falls_back_when_db_empty(item, mocker): + mocker.patch.object(fn_main, "execute_query", return_value=None) + + assert fn_main.get_latest_file_naming_status(item) == FileNamingStatus.PROCESSING + + +def test_get_latest_status_falls_back_on_unknown_value(item, mocker): + mocker.patch.object(fn_main, "execute_query", return_value="NOT_A_REAL_STATUS") + + assert fn_main.get_latest_file_naming_status(item) == FileNamingStatus.PROCESSING + + +def test_callback_with_non_processitem_does_not_crash(mocker): + """A message that does not deserialize to a ProcessItem must be acknowledged + and skipped without touching the database or forwarding to the next queue.""" + execute_query = mocker.patch.object(fn_main, "execute_query") + forward = mocker.patch.object(fn_main, "forward_to_rabbitmq") + mocker.patch.object(fn_main, "update_scanneddata_database") + + ch = mocker.Mock() + method = mocker.Mock() + method.delivery_tag = 123 + body = pickle.dumps({"not": "a process item"}) + + fn_main.callback(ch, method, None, body) + + ch.basic_ack.assert_called_once_with(delivery_tag=123) + execute_query.assert_not_called() + forward.assert_not_called() diff --git a/tests/test_homepage.py b/tests/test_homepage.py index 8e2db24..0f8ebf2 100644 --- a/tests/test_homepage.py +++ b/tests/test_homepage.py @@ -7,6 +7,44 @@ from selenium.webdriver.common.by import By +def _render_progress_segments(driver, pdf_data): + script = """ + const pdfData = arguments[0]; + const existingCol = document.getElementById(pdfData.id + '_col'); + if (existingCol) { + existingCol.remove(); + } + addPdfCard(pdfData); + const progressBar = document.getElementById(pdfData.id + '_progress_bar'); + return Array.from(progressBar.querySelectorAll('.progress-segment')).map((segment) => ({ + classes: Array.from(segment.classList), + tooltip: segment.getAttribute('data-tooltip'), + ariaLabel: segment.getAttribute('aria-label'), + })); + """ + return driver.execute_script(script, pdf_data) + + +def _update_progress_segments(driver, initial_pdf_data, update_data): + script = """ + const initialPdfData = arguments[0]; + const updateData = arguments[1]; + const existingCol = document.getElementById(initialPdfData.id + '_col'); + if (existingCol) { + existingCol.remove(); + } + addPdfCard(initialPdfData); + updateCard(updateData); + const progressBar = document.getElementById(initialPdfData.id + '_progress_bar'); + return Array.from(progressBar.querySelectorAll('.progress-segment')).map((segment) => ({ + classes: Array.from(segment.classList), + tooltip: segment.getAttribute('data-tooltip'), + ariaLabel: segment.getAttribute('aria-label'), + })); + """ + return driver.execute_script(script, initial_pdf_data, update_data) + + @pytest.fixture def driver(): options = Options() @@ -124,3 +162,252 @@ def test_dashboard_settings_ollama_first_start(driver): error_div = driver.find_element(By.ID, "ollama-error") models_section = driver.find_element(By.ID, "ollama-models-section") assert error_div.is_displayed() or models_section.is_displayed() + + +@pytest.mark.parametrize( + "pdf_data,expected", + [ + ( + { + "id": 9900, + "file_name": "metadata-default.pdf", + "file_status": "Reading Metadata", + "status_progressbar": None, + "badges": [], + }, + [ + ("completed", "File Detection โ€“ Completed"), + ("current", "Reading Metadata โ€“ In Progress"), + ("pending", "OCR โ€“ Pending"), + ("pending", "File Naming โ€“ Pending"), + ("pending", "Upload โ€“ Pending"), + ], + ), + ( + { + "id": 9901, + "file_name": "progress-step-2.pdf", + "file_status": "OCR Processing", + "status_progressbar": 2, + "badges": [], + }, + [ + ("completed", "File Detection โ€“ Completed"), + ("completed", "Reading Metadata โ€“ Completed"), + ("current", "OCR โ€“ In Progress"), + ("pending", "File Naming โ€“ Pending"), + ("pending", "Upload โ€“ Pending"), + ], + ), + ( + { + "id": 9902, + "file_name": "invalid-file.pdf", + "file_status": "Invalid File", + "status_progressbar": -1, + "badges": [], + }, + [ + ("failed", "File Detection โ€“ Failed"), + ("pending", "Reading Metadata โ€“ Pending"), + ("pending", "OCR โ€“ Pending"), + ("pending", "File Naming โ€“ Pending"), + ("pending", "Upload โ€“ Pending"), + ], + ), + ( + { + "id": 9903, + "file_name": "ocr-failed.pdf", + "file_status": "Failed", + "status_progressbar": 3, + "ocr_status": "INPUT_ERROR", + "badges": [], + }, + [ + ("completed", "File Detection โ€“ Completed"), + ("completed", "Reading Metadata โ€“ Completed"), + ("failed", "OCR โ€“ Failed"), + ("pending", "File Naming โ€“ Pending"), + ("pending", "Upload โ€“ Pending"), + ], + ), + ( + { + "id": 9904, + "file_name": "naming-failed.pdf", + "file_status": "Failed", + "status_progressbar": 4, + "file_naming_status": "NO_SERVER_CONNECTION", + "badges": [], + }, + [ + ("completed", "File Detection โ€“ Completed"), + ("completed", "Reading Metadata โ€“ Completed"), + ("completed", "OCR โ€“ Completed"), + ("failed", "File Naming โ€“ Failed"), + ("pending", "Upload โ€“ Pending"), + ], + ), + ( + { + "id": 9905, + "file_name": "sync-failed.pdf", + "file_status": "Sync Failed", + "status_progressbar": 5, + "badges": [], + }, + [ + ("completed", "File Detection โ€“ Completed"), + ("completed", "Reading Metadata โ€“ Completed"), + ("completed", "OCR โ€“ Completed"), + ("completed", "File Naming โ€“ Completed"), + ("failed", "Upload โ€“ Failed"), + ], + ), + ( + { + "id": 9906, + "file_name": "deleted.pdf", + "file_status": "Deleted", + "status_progressbar": 2, + "badges": [], + }, + [ + ("failed", "File Detection โ€“ Failed"), + ("failed", "Reading Metadata โ€“ Failed"), + ("failed", "OCR โ€“ Failed"), + ("failed", "File Naming โ€“ Failed"), + ("failed", "Upload โ€“ Failed"), + ], + ), + ( + { + "id": 9908, + "file_name": "file-detection.pdf", + "file_status": "File Not Ready", + "status_progressbar": 0, + "badges": [], + }, + [ + ("current", "File Detection โ€“ In Progress"), + ("pending", "Reading Metadata โ€“ Pending"), + ("pending", "OCR โ€“ Pending"), + ("pending", "File Naming โ€“ Pending"), + ("pending", "Upload โ€“ Pending"), + ], + ), + ( + { + "id": 9907, + "file_name": "completed-with-ocr-failure.pdf", + "file_status": "Completed", + "status_progressbar": 5, + "ocr_status": "UNSUPPORTED", + "badges": [], + }, + [ + ("completed", "File Detection โ€“ Completed"), + ("completed", "Reading Metadata โ€“ Completed"), + ("failed", "OCR โ€“ Failed"), + ("completed", "File Naming โ€“ Completed"), + ("completed", "Upload โ€“ Completed"), + ], + ), + ], +) +def test_dashboard_progress_bar_step_statuses(driver, pdf_data, expected): + driver.get("http://web-service:5001") + WebDriverWait(driver, 10).until(EC.title_contains("ScanSync")) + + rendered = _render_progress_segments(driver, pdf_data) + assert len(rendered) == 5 + + for index, (expected_class, expected_tooltip) in enumerate(expected): + segment = rendered[index] + assert expected_class in segment["classes"] + assert segment["tooltip"] == expected_tooltip + assert segment["ariaLabel"] == expected_tooltip + + +@pytest.mark.parametrize( + "initial_pdf_data,update_data,expected", + [ + ( + { + "id": 9911, + "file_name": "step-zero-update.pdf", + "file_status": "Reading Metadata", + "status_progressbar": 1, + "badges": [], + }, + { + "id": 9911, + "file_status": "File Not Ready", + "status_progressbar": 0, + }, + [ + ("current", "File Detection โ€“ In Progress"), + ("pending", "Reading Metadata โ€“ Pending"), + ("pending", "OCR โ€“ Pending"), + ("pending", "File Naming โ€“ Pending"), + ("pending", "Upload โ€“ Pending"), + ], + ), + ( + { + "id": 9912, + "file_name": "generic-failure.pdf", + "file_status": "OCR Processing", + "status_progressbar": 2, + "badges": [], + }, + { + "id": 9912, + "file_status": "Failed", + "status_progressbar": "-1", + }, + [ + ("failed", "File Detection โ€“ Failed"), + ("failed", "Reading Metadata โ€“ Failed"), + ("failed", "OCR โ€“ Failed"), + ("failed", "File Naming โ€“ Failed"), + ("failed", "Upload โ€“ Failed"), + ], + ), + ( + { + "id": 9913, + "file_name": "live-file-naming.pdf", + "file_status": "File Name Processing", + "status_progressbar": 3, + "badges": [], + }, + { + "id": 9913, + "file_status": "Failed", + "status_progressbar": 4, + "file_naming_status": "NO_SERVER_CONNECTION", + }, + [ + ("completed", "File Detection โ€“ Completed"), + ("completed", "Reading Metadata โ€“ Completed"), + ("completed", "OCR โ€“ Completed"), + ("failed", "File Naming โ€“ Failed"), + ("pending", "Upload โ€“ Pending"), + ], + ), + ], +) +def test_dashboard_progress_bar_live_updates(driver, initial_pdf_data, update_data, expected): + driver.get("http://web-service:5001") + WebDriverWait(driver, 10).until(EC.title_contains("ScanSync")) + + rendered = _update_progress_segments(driver, initial_pdf_data, update_data) + assert len(rendered) == 5 + + for index, (expected_class, expected_tooltip) in enumerate(expected): + segment = rendered[index] + assert expected_class in segment["classes"] + assert segment["tooltip"] == expected_tooltip + assert segment["ariaLabel"] == expected_tooltip diff --git a/tests/test_ocr_job_tracking.py b/tests/test_ocr_job_tracking.py new file mode 100644 index 0000000..6fc010b --- /dev/null +++ b/tests/test_ocr_job_tracking.py @@ -0,0 +1,186 @@ +import sys +import types + +import pytest + + +# ocrmypdf is a heavy native dependency that is not installed in the test +# environment. Provide a lightweight stub so the OCR service module can be +# imported and its job-tracking logic exercised in isolation. +_ocrmypdf_stub = types.ModuleType("ocrmypdf") + + +class _UnsupportedImageFormatError(Exception): + pass + + +class _DpiError(Exception): + pass + + +class _InputFileError(Exception): + pass + + +class _OutputFileAccessError(Exception): + pass + + +class _MissingDependencyError(Exception): + pass + + +_ocrmypdf_stub.UnsupportedImageFormatError = _UnsupportedImageFormatError +_ocrmypdf_stub.DpiError = _DpiError +_ocrmypdf_stub.InputFileError = _InputFileError +_ocrmypdf_stub.OutputFileAccessError = _OutputFileAccessError +_ocrmypdf_stub.MissingDependencyError = _MissingDependencyError +_ocrmypdf_stub.ocr = lambda *args, **kwargs: 0 + +# Importing the service pulls in scansynclib.sqlite_wrapper, which initializes a +# real SQLite database at import time. That database is not available in the unit +# test environment, so replace the module with a stub. The query helpers are +# mocked per-test anyway. +_sqlite_stub = types.ModuleType("scansynclib.sqlite_wrapper") +_sqlite_stub.execute_query = lambda *args, **kwargs: None +_sqlite_stub.update_scanneddata_database = lambda *args, **kwargs: None + +_original_modules = { + "ocrmypdf": sys.modules.get("ocrmypdf"), + "scansynclib.sqlite_wrapper": sys.modules.get("scansynclib.sqlite_wrapper"), +} +sys.modules["ocrmypdf"] = _ocrmypdf_stub +sys.modules["scansynclib.sqlite_wrapper"] = _sqlite_stub + + +def teardown_module(module): + for name, original in _original_modules.items(): + if original is None: + sys.modules.pop(name, None) + else: + sys.modules[name] = original + + +import ocr_service.main as ocr_main # noqa: E402 +from scansynclib.ProcessItem import ProcessItem, ItemType, OCRStatus, ProcessStatus # noqa: E402 + + +@pytest.fixture +def item(tmp_path): + file_path = tmp_path / "scan.pdf" + file_path.write_bytes(b"%PDF-1.4 test") + process_item = ProcessItem(str(file_path), ItemType.PDF) + process_item.db_id = 42 + return process_item + + +@pytest.fixture +def patched(mocker): + """Mock external collaborators of the OCR service. + + execute_query returns the same fake row id (99) for the INSERT, which the + service stores as ocr_db_id and reuses for the UPDATE. File naming is + disabled by default so processed items are forwarded to the upload queue. + """ + execute_query = mocker.patch.object(ocr_main, "execute_query", return_value=99) + update_db = mocker.patch.object(ocr_main, "update_scanneddata_database") + forward = mocker.patch.object(ocr_main, "forward_to_rabbitmq") + fake_settings = types.SimpleNamespace( + file_naming=types.SimpleNamespace( + ollama_server_url="", + ollama_server_port="", + ollama_model="", + openai_api_key="", + ) + ) + mocker.patch.object(ocr_main, "settings", fake_settings) + return { + "execute_query": execute_query, + "update_db": update_db, + "forward": forward, + "settings": fake_settings, + } + + +def _ocr_job_update_args(execute_query): + update_calls = [c for c in execute_query.call_args_list if "UPDATE ocr_jobs" in c.args[0]] + assert len(update_calls) == 1, "Expected exactly one ocr_jobs status update" + return update_calls[0].args[1] + + +@pytest.mark.parametrize( + "ocr_behavior, expected_status, expected_error", + [ + ({"return_value": 0}, OCRStatus.COMPLETED, None), + ({"return_value": 5}, OCRStatus.FAILED, "OCR exited with code 5"), + ( + {"side_effect": ocr_main.ocrmypdf.UnsupportedImageFormatError()}, + OCRStatus.UNSUPPORTED, + "Unsupported image format", + ), + ( + {"side_effect": ocr_main.ocrmypdf.DpiError("dpi too low")}, + OCRStatus.DPI_ERROR, + "dpi too low", + ), + ( + {"side_effect": ocr_main.ocrmypdf.InputFileError("bad input")}, + OCRStatus.INPUT_ERROR, + "bad input", + ), + ( + {"side_effect": ocr_main.ocrmypdf.OutputFileAccessError("no write access")}, + OCRStatus.OUTPUT_ERROR, + "no write access", + ), + ( + {"side_effect": ocr_main.ocrmypdf.MissingDependencyError()}, + OCRStatus.FAILED, + "Missing OCR dependency", + ), + ({"side_effect": ValueError("boom")}, OCRStatus.FAILED, "boom"), + ], +) +def test_start_processing_persists_ocr_job_status(item, patched, mocker, ocr_behavior, expected_status, expected_error): + mocker.patch.object(ocr_main.ocrmypdf, "ocr", **ocr_behavior) + # On the success path the service verifies that ocrmypdf produced an output + # file containing text. ocrmypdf is stubbed here, so provide the artifacts it + # would normally create for the "completed" case. + with open(item.ocr_file, "wb") as ocr_file: + ocr_file.write(b"%PDF-1.4 ocr") + mocker.patch.object(ocr_main, "extract_text", return_value="sample text") + + ocr_main.start_processing(item) + + assert item.ocr_status == expected_status + + status_name, error, db_id = _ocr_job_update_args(patched["execute_query"]) + assert status_name == expected_status.name + assert error == expected_error + assert db_id == 99 + + +def test_start_processing_inserts_ocr_job_and_forwards_to_upload(item, patched, mocker): + mocker.patch.object(ocr_main.ocrmypdf, "ocr", return_value=0) + + ocr_main.start_processing(item) + + insert_calls = [c for c in patched["execute_query"].call_args_list if "INSERT INTO ocr_jobs" in c.args[0]] + assert len(insert_calls) == 1 + assert insert_calls[0].args[1] == (42, OCRStatus.PROCESSING.name) + assert insert_calls[0].kwargs.get("return_last_id") is True + + patched["forward"].assert_called_once() + assert patched["forward"].call_args.args[0] == "upload_queue" + assert item.status == ProcessStatus.SYNC_PENDING + + +def test_start_processing_forwards_to_file_naming_when_enabled(item, patched, mocker): + patched["settings"].file_naming.openai_api_key = "secret" + mocker.patch.object(ocr_main.ocrmypdf, "ocr", return_value=0) + + ocr_main.start_processing(item) + + patched["forward"].assert_called_once() + assert patched["forward"].call_args.args[0] == "file_naming_queue" + assert item.status == ProcessStatus.FILENAME_PENDING diff --git a/tests/test_ocr_verification.py b/tests/test_ocr_verification.py index c67f584..397dd4c 100644 --- a/tests/test_ocr_verification.py +++ b/tests/test_ocr_verification.py @@ -188,8 +188,8 @@ def test_ocr_success_with_text_sets_completed(self): final_call = mock_update_db.call_args_list[-1][0][1] assert final_call.get("ocr_status") == OCRStatus.COMPLETED.name - def test_ocr_success_no_text_sets_failed(self): - """When OCR succeeds but no text found, ocr_status should be FAILED.""" + def test_ocr_success_no_text_sets_no_text(self): + """When OCR succeeds but no text found, ocr_status should be NO_TEXT.""" ocr_main, mock_ocrmypdf = _load_ocr_main() with patch.object(ocr_main, 'ocrmypdf') as mock_ocr_mod, \ @@ -205,12 +205,12 @@ def test_ocr_success_no_text_sets_failed(self): item = self._create_mock_item() ocr_main.start_processing(item) - assert item.ocr_status == OCRStatus.FAILED + assert item.ocr_status == OCRStatus.NO_TEXT final_call = mock_update_db.call_args_list[-1][0][1] - assert final_call.get("ocr_status") == OCRStatus.FAILED.name + assert final_call.get("ocr_status") == OCRStatus.NO_TEXT.name - def test_ocr_success_whitespace_only_sets_failed(self): - """When OCR succeeds but only whitespace found, ocr_status should be FAILED.""" + def test_ocr_success_whitespace_only_sets_no_text(self): + """When OCR succeeds but only whitespace found, ocr_status should be NO_TEXT.""" ocr_main, mock_ocrmypdf = _load_ocr_main() with patch.object(ocr_main, 'ocrmypdf') as mock_ocr_mod, \ @@ -226,7 +226,7 @@ def test_ocr_success_whitespace_only_sets_failed(self): item = self._create_mock_item() ocr_main.start_processing(item) - assert item.ocr_status == OCRStatus.FAILED + assert item.ocr_status == OCRStatus.NO_TEXT def test_ocr_success_missing_output_file_sets_output_error(self): """When OCR succeeds but output file is missing, status should be OUTPUT_ERROR.""" diff --git a/web_service/src/main.py b/web_service/src/main.py index b3a870e..cb88b0a 100644 --- a/web_service/src/main.py +++ b/web_service/src/main.py @@ -94,6 +94,7 @@ def callback(ch, method, properties, body): current_upload_target=item.current_upload_target, badges=badges, # Add the generated badges ocr_status=item.ocr_status.name if getattr(item, "ocr_status", None) else None, + file_naming_status=item.file_naming_status.name if getattr(item, "file_naming_status", None) else None, ) payload["dashboard_data"] = get_dashboard_info() # Nur bei Bedarf abrufen sse_queue.put(json.dumps(payload, default=str)) # Ensure all objects are serializable diff --git a/web_service/src/routes/dashboard.py b/web_service/src/routes/dashboard.py index 8baecee..10109c3 100644 --- a/web_service/src/routes/dashboard.py +++ b/web_service/src/routes/dashboard.py @@ -43,7 +43,8 @@ def index(): stats.processing_pdfs, stats.latest_processing, stats.latest_completed, - smb.id AS smb_target_id + smb.id AS smb_target_id, + fn.file_naming_status AS file_naming_status FROM ( SELECT COUNT(*) AS total_entries, @@ -64,6 +65,12 @@ def index(): LIMIT :limit OFFSET :offset ) d ON 1=1 LEFT JOIN smb_onedrive smb ON d.local_filepath = smb.smb_name + LEFT JOIN file_naming_jobs fn + ON fn.id = ( + SELECT MAX(id) + FROM file_naming_jobs + WHERE scanneddata_id = d.id + ) ''' result = db.execute(query, {'limit': entries_per_page, 'offset': offset}).fetchall() diff --git a/web_service/src/static/css/dashboard.css b/web_service/src/static/css/dashboard.css index 569e509..479ede9 100644 --- a/web_service/src/static/css/dashboard.css +++ b/web_service/src/static/css/dashboard.css @@ -44,7 +44,7 @@ height: 8px; background-color: #e9ecef; border-radius: 4px; - overflow: hidden; + overflow: visible; gap: 2px; margin-bottom: 1px; } @@ -54,18 +54,53 @@ flex: 1; background-color: #dee2e6; transition: background-color 0.3s; border-radius: 2px; +position: relative; +cursor: default; +} + +.progress-segment.completed { +background-color: #198754; } -.progress-segment.active { -background-color: #0d6efd; /* Bootstrap-Blau */ +.progress-segment.current { +background: linear-gradient(90deg, #0d6efd 0%, #6db3f8 50%, #0d6efd 100%); +background-size: 200% 100%; +animation: marquee 1.5s linear infinite; } .progress-segment.failed { -background-color: #dc3545; /* Bootstrap-Rot */ +background-color: #dc3545; } -.progress-segment.completed { -background-color: green; /* Bootstrap-Rot */ +.progress-segment.pending { +background-color: #dee2e6; +} + +@keyframes marquee { +0% { background-position: 100% 0; } +100% { background-position: -100% 0; } +} + +.progress-segment[data-tooltip]:hover::after, +.progress-segment[data-tooltip]:focus-visible::after { + content: attr(data-tooltip); + position: absolute; + bottom: calc(100% + 6px); + left: 50%; + transform: translateX(-50%); + background-color: rgba(0, 0, 0, 0.85); + color: #fff; + padding: 4px 8px; + border-radius: 4px; + font-size: 0.7rem; + white-space: nowrap; + z-index: 10; + pointer-events: none; +} + +.progress-segment:focus-visible { + outline: 2px solid #0d6efd; + outline-offset: 1px; } .rotating { diff --git a/web_service/src/static/js/dashboard.js b/web_service/src/static/js/dashboard.js index 97a5ae9..ca59427 100644 --- a/web_service/src/static/js/dashboard.js +++ b/web_service/src/static/js/dashboard.js @@ -229,10 +229,10 @@ function updateCard(updateData) { // Update Progress Bar try { - if (updateData.status_progressbar) { + if (updateData.status_progressbar !== undefined && updateData.status_progressbar !== null) { const progressBar = document.getElementById(`${updateData.id}_progress_bar`); if (progressBar) { - updateProgressBar(updateData.id, updateData.status_progressbar); + updateProgressBar(updateData.id, updateData.status_progressbar, updateData); } else { console.warn(`Progress bar with ID ${updateData.id}_progress_bar not found.`); } @@ -492,20 +492,38 @@ function addPdfCard(pdfData) { progressContainer.classList.add('progress-bar-wrapper'); progressContainer.id = pdfData.id + '_progress_bar'; - const progressStep = pdfData.status_progressbar || 1; - const isFailed = pdfData.file_status?.toLowerCase().includes("failed") || pdfData.file_status?.toLowerCase().includes("deleted") || pdfData.status_progressbar === -1; + const hasProgressStep = pdfData.status_progressbar !== undefined && pdfData.status_progressbar !== null && pdfData.status_progressbar !== ""; + const parsedProgressStep = hasProgressStep ? Number(pdfData.status_progressbar) : NaN; + const progressStep = Number.isFinite(parsedProgressStep) ? parsedProgressStep : 1; + const isDeleted = pdfData.file_status?.toLowerCase().includes("deleted"); + const isFailed = pdfData.file_status?.toLowerCase().includes("failed") || pdfData.file_status?.toLowerCase().includes("invalid file") || isDeleted || progressStep === -1; const isCompleted = pdfData.file_status?.toLowerCase().includes("completed"); + const stepLabels = ["File Detection", "Reading Metadata", "OCR", "File Naming", "Upload"]; + const stepStatuses = getStepStatuses(progressStep, isFailed, isDeleted, isCompleted, pdfData); for (let i = 0; i < 5; i++) { const segment = document.createElement('div'); segment.classList.add('progress-segment'); + segment.setAttribute('tabindex', '0'); + segment.setAttribute('role', 'img'); - if (isFailed) { + const status = stepStatuses[i]; + if (status === "failed") { segment.classList.add('failed'); - } else if (isCompleted) { + segment.setAttribute('data-tooltip', stepLabels[i] + " โ€“ Failed"); + segment.setAttribute('aria-label', stepLabels[i] + " โ€“ Failed"); + } else if (status === "completed") { segment.classList.add('completed'); - } else if (i < progressStep) { - segment.classList.add('active'); + segment.setAttribute('data-tooltip', stepLabels[i] + " โ€“ Completed"); + segment.setAttribute('aria-label', stepLabels[i] + " โ€“ Completed"); + } else if (status === "current") { + segment.classList.add('current'); + segment.setAttribute('data-tooltip', stepLabels[i] + " โ€“ In Progress"); + segment.setAttribute('aria-label', stepLabels[i] + " โ€“ In Progress"); + } else { + segment.classList.add('pending'); + segment.setAttribute('data-tooltip', stepLabels[i] + " โ€“ Pending"); + segment.setAttribute('aria-label', stepLabels[i] + " โ€“ Pending"); } progressContainer.appendChild(segment); @@ -575,9 +593,12 @@ function getStatusIcon(file_status) { return status_icon; } +// Human-readable warning text for a failed OCR status. Returns null for +// non-failure statuses so the OCR warning is only rendered when OCR failed. function getOcrStatusText(ocr_status) { const ocrFailureMessages = { 'FAILED': 'OCR: Failed', + 'NO_TEXT': 'OCR: No text found', 'UNSUPPORTED': 'OCR: Unsupported format', 'DPI_ERROR': 'OCR: Image DPI too low', 'INPUT_ERROR': 'OCR: Input file error', @@ -586,24 +607,146 @@ function getOcrStatusText(ocr_status) { return ocrFailureMessages[ocr_status] || null; } -function updateProgressBar(pdfId, newStep) { +// OCR statuses that indicate failure +const ocrFailureStatuses = ["FAILED", "NO_TEXT", "UNSUPPORTED", "DPI_ERROR", "INPUT_ERROR", "OUTPUT_ERROR"]; + +// File naming statuses that indicate failure +const fileNamingFailureStatuses = ["FAILED", "NO_OCR_FILE", "NO_PDF_TEXT", "NO_SERVER_CONNECTION", + "MODEL_NOT_FOUND", "AUTHENTICATION_ERROR", "RATE_LIMIT_ERROR"]; + +/** + * Resolve which of the 5 progress steps is currently active from the textual + * file_status. The numeric status_progressbar maps "pending" states to the + * previously completed step (e.g. "File Name Pending" -> 2 = OCR), which would + * place the in-progress marker on the wrong/failed step. Deriving it from the + * status text keeps the active marker (and its marquee animation) on the stage + * the pipeline has actually advanced to. + */ +function getCurrentStepIndex(fileStatus, progressStep) { + const status = (fileStatus || "").toLowerCase(); + if (status.includes("file name") || status.includes("file naming")) return 3; + if (status.includes("sync") || status.includes("upload")) return 4; + if (status.includes("ocr")) return 2; + if (status.includes("metadata")) return 1; + if (status.includes("not ready") || status.includes("detection")) return 0; + return progressStep; +} + +/** + * Determine the visual status for each of the 5 progress bar steps. + * Returns an array of 5 strings: "completed", "failed", "current", or "pending". + */ +function getStepStatuses(progressStep, isFailed, isDeleted, isCompleted, pdfData) { + const statuses = ["pending", "pending", "pending", "pending", "pending"]; + + // If file is deleted, all steps show as failed + if (isDeleted) { + return ["failed", "failed", "failed", "failed", "failed"]; + } + + // If overall failed, mark completed steps and the failing step + if (isFailed) { + // Determine which step failed based on file_status and per-step statuses + const fileStatus = (pdfData.file_status || "").toLowerCase(); + let failedStep = -1; + + if (fileStatus.includes("invalid file")) { + failedStep = 0; // Detection failed + } else if (fileStatus === "failed" && progressStep >= 0 && progressStep <= 1) { + failedStep = 1; // Metadata failed + } else if (pdfData.ocr_status && ocrFailureStatuses.includes(pdfData.ocr_status)) { + failedStep = 2; // OCR failed + } else if (pdfData.file_naming_status && fileNamingFailureStatuses.includes(pdfData.file_naming_status)) { + failedStep = 3; // File naming failed + } else if (fileStatus.includes("sync failed")) { + failedStep = 4; // Upload failed + } else { + // Generic failure - mark all as failed + return ["failed", "failed", "failed", "failed", "failed"]; + } + + for (let i = 0; i < 5; i++) { + if (i < failedStep) { + statuses[i] = "completed"; + } else if (i === failedStep) { + statuses[i] = "failed"; + } + } + return statuses; + } + + // All completed + if (isCompleted) { + for (let i = 0; i < 5; i++) { + // Check if individual steps had issues even though overall completed + if (i === 2 && pdfData.ocr_status && ocrFailureStatuses.includes(pdfData.ocr_status)) { + statuses[i] = "failed"; + } else if (i === 3 && pdfData.file_naming_status && fileNamingFailureStatuses.includes(pdfData.file_naming_status)) { + statuses[i] = "failed"; + } else { + statuses[i] = "completed"; + } + } + return statuses; + } + + // In progress - mark completed steps, current step, and check per-step failures + const currentStep = getCurrentStepIndex(pdfData.file_status, progressStep); + for (let i = 0; i < 5; i++) { + // Surface OCR / file naming failures regardless of the current step. After a + // non-fatal OCR failure processing continues to the next stage, so the OCR step + // can equal progressStep while it has already failed. Checking the sub-status + // first prevents showing a failed step as still "in progress". + if (i === 2 && pdfData.ocr_status && ocrFailureStatuses.includes(pdfData.ocr_status)) { + statuses[i] = "failed"; + } else if (i === 3 && pdfData.file_naming_status && fileNamingFailureStatuses.includes(pdfData.file_naming_status)) { + statuses[i] = "failed"; + } else if (i < currentStep) { + statuses[i] = "completed"; + } else if (i === currentStep) { + statuses[i] = "current"; + } + // else remains "pending" + } + return statuses; +} + +function updateProgressBar(pdfId, newStep, pdfData) { const progressBar = document.getElementById(`${pdfId}_progress_bar`); if (!progressBar) return; const segments = progressBar.querySelectorAll('.progress-segment'); + const stepLabels = ["File Detection", "Reading Metadata", "OCR", "File Naming", "Upload"]; // Clamp value to -1โ€“5 const clampedStep = Math.max(-1, Math.min(5, newStep)); + const fileStatus = pdfData?.file_status?.toLowerCase() || ""; + const isDeleted = fileStatus.includes("deleted"); + const isFailed = fileStatus.includes("failed") || fileStatus.includes("invalid file") || isDeleted || clampedStep === -1; + const isCompleted = fileStatus.includes("completed") || clampedStep === 5; + const stepStatuses = getStepStatuses(clampedStep, isFailed, isDeleted, isCompleted, pdfData || {}); + segments.forEach((segment, index) => { - segment.classList.remove('active', 'failed', 'completed'); + segment.classList.remove('active', 'failed', 'completed', 'current', 'pending'); - if (clampedStep === -1) { + const status = stepStatuses[index]; + if (status === "failed") { segment.classList.add('failed'); - } else if (clampedStep === 5) { + segment.setAttribute('data-tooltip', stepLabels[index] + " โ€“ Failed"); + segment.setAttribute('aria-label', stepLabels[index] + " โ€“ Failed"); + } else if (status === "completed") { segment.classList.add('completed'); - } else if (index < clampedStep) { - segment.classList.add('active'); + segment.setAttribute('data-tooltip', stepLabels[index] + " โ€“ Completed"); + segment.setAttribute('aria-label', stepLabels[index] + " โ€“ Completed"); + } else if (status === "current") { + segment.classList.add('current'); + segment.setAttribute('data-tooltip', stepLabels[index] + " โ€“ In Progress"); + segment.setAttribute('aria-label', stepLabels[index] + " โ€“ In Progress"); + } else { + segment.classList.add('pending'); + segment.setAttribute('data-tooltip', stepLabels[index] + " โ€“ Pending"); + segment.setAttribute('aria-label', stepLabels[index] + " โ€“ Pending"); } }); } \ No newline at end of file