diff --git a/CRITICAL_ISSUES_REPORT.md b/CRITICAL_ISSUES_REPORT.md index 78af4bcb..ee3b276f 100644 --- a/CRITICAL_ISSUES_REPORT.md +++ b/CRITICAL_ISSUES_REPORT.md @@ -270,15 +270,15 @@ def update_contest(contest_id): ### Immediate (Before Any Production Deployment) -1. ✅ **Fix hardcoded secrets** - Remove `'rohank10'` defaults -2. ✅ **Secure debug endpoint** - Add authentication or remove -3. ✅ **Revoke OAuth credentials** - Generate new ones, move to env vars -4. ✅ **Disable debug mode** - Use environment variable +1. **Fix hardcoded secrets** - Remove `'rohank10'` defaults +2. **Secure debug endpoint** - Add authentication or remove +3. **Revoke OAuth credentials** - Generate new ones, move to env vars +4. **Disable debug mode** - Use environment variable ### High Priority (Before Next Release) -5. ✅ **Fix database password default** - Require DATABASE_URL -6. ✅ **Add error handler** - Fix update_contest route +5. **Fix database password default** - Require DATABASE_URL +6. **Add error handler** - Fix update_contest route ### Additional Recommendations diff --git a/backend/alembic/versions/0f85a2313fbd_add_automated_settings_column.py b/backend/alembic/versions/0f85a2313fbd_add_automated_settings_column.py new file mode 100644 index 00000000..3a6407c5 --- /dev/null +++ b/backend/alembic/versions/0f85a2313fbd_add_automated_settings_column.py @@ -0,0 +1,59 @@ +"""add_automated_settings_column + +Revision ID: 0f85a2313fbd +Revises: 2b7c1a9d4c3e +Create Date: 2026-03-17 + +Adds automated_settings column to contests table for automated scoring mode. +This column stores eligibility criteria and evaluation parameters as JSON. +""" + +from alembic import op +import sqlalchemy as sa +from sqlalchemy.dialects import mysql + +# revision identifiers, used by Alembic. +revision = '0f85a2313fbd' +down_revision = '2b7c1a9d4c3e' +branch_labels = None +depends_on = None + + +def upgrade(): + """ + Add automated_settings column to contests table. + + Structure stored as JSON: + { + "enabled": true, + "eligibility": { + "min_edits": 100, + "min_outgoing_links": 3 + }, + "evaluation": { + "points_per_accepted": 10, + "points_per_byte": 0.001, + "points_per_incoming_link": 2, + "points_per_outgoing_link": 1, + "points_per_category": 1, + "points_per_new_reference": 3, + "points_per_reused_reference": 1, + "points_per_infobox": 5, + "points_per_image": 2 + } + } + """ + op.add_column( + 'contests', + sa.Column( + 'automated_settings', + sa.Text(), + nullable=True, + comment='Automated scoring configuration (eligibility + evaluation criteria)' + ) + ) + + +def downgrade(): + """Remove automated_settings column from contests table.""" + op.drop_column('contests', 'automated_settings') diff --git a/backend/alembic/versions/195a1fb1d56e_add_incoming_and_outgoing_links_columns.py b/backend/alembic/versions/195a1fb1d56e_add_incoming_and_outgoing_links_columns.py new file mode 100644 index 00000000..67e783b8 --- /dev/null +++ b/backend/alembic/versions/195a1fb1d56e_add_incoming_and_outgoing_links_columns.py @@ -0,0 +1,46 @@ +"""add incoming and outgoing links columns + +Revision ID: 195a1fb1d56e +Revises: f1a2b3c4d5e6 +Create Date: 2026-02-06 17:25:37.027375 + +""" +from alembic import op +import sqlalchemy as sa +from sqlalchemy import inspect + +# revision identifiers, used by Alembic. +revision = '195a1fb1d56e' +down_revision = "f1a2b3c4d5e6" +branch_labels = None +depends_on = None + + +def upgrade() -> None: + conn = op.get_bind() + inspector = inspect(conn) + columns = [col["name"] for col in inspector.get_columns("submissions")] + + if "incoming_links" not in columns: + op.add_column( + "submissions", + sa.Column("incoming_links", sa.Integer(), nullable=True), + ) + + if "outgoing_links" not in columns: + op.add_column( + "submissions", + sa.Column("outgoing_links", sa.Integer(), nullable=True), + ) + + +def downgrade() -> None: + conn = op.get_bind() + inspector = inspect(conn) + columns = [col["name"] for col in inspector.get_columns("submissions")] + + if "outgoing_links" in columns: + op.drop_column("submissions", "outgoing_links") + + if "incoming_links" in columns: + op.drop_column("submissions", "incoming_links") diff --git a/backend/alembic/versions/1a2b3c4d5e6f_add_evaluation_details_columns.py b/backend/alembic/versions/1a2b3c4d5e6f_add_evaluation_details_columns.py new file mode 100644 index 00000000..d6d44037 --- /dev/null +++ b/backend/alembic/versions/1a2b3c4d5e6f_add_evaluation_details_columns.py @@ -0,0 +1,46 @@ +"""add_evaluation_details_columns + +Revision ID: 1a2b3c4d5e6f +Revises: 0f85a2313fbd +Create Date: 2026-03-21 + +Adds evaluation_reason and score_breakdown columns to submissions table +for displaying automated scoring details. +""" + +from alembic import op +import sqlalchemy as sa + +# revision identifiers, used by Alembic. +revision = '1a2b3c4d5e6f' +down_revision = '0f85a2313fbd' +branch_labels = None +depends_on = None + + +def upgrade(): + """Add evaluation_reason and score_breakdown columns to submissions table.""" + op.add_column( + 'submissions', + sa.Column( + 'evaluation_reason', + sa.Text(), + nullable=True, + comment='Reason for rejection or success message from automated evaluation' + ) + ) + op.add_column( + 'submissions', + sa.Column( + 'score_breakdown', + sa.Text(), + nullable=True, + comment='JSON breakdown of score calculation for automated scoring' + ) + ) + + +def downgrade(): + """Remove evaluation_reason and score_breakdown columns from submissions table.""" + op.drop_column('submissions', 'score_breakdown') + op.drop_column('submissions', 'evaluation_reason') diff --git a/backend/alembic/versions/2b7c1a9d4c3e_add_detailed_reference_counts_to_submissions.py b/backend/alembic/versions/2b7c1a9d4c3e_add_detailed_reference_counts_to_submissions.py new file mode 100644 index 00000000..c961af7e --- /dev/null +++ b/backend/alembic/versions/2b7c1a9d4c3e_add_detailed_reference_counts_to_submissions.py @@ -0,0 +1,46 @@ +"""Add ref_new_count and ref_reused_count columns to submissions table + +Revision ID: 2b7c1a9d4c3e +Revises: 195a1fb1d56e + +""" + +from alembic import op +import sqlalchemy as sa +from sqlalchemy import inspect + + +revision = "2b7c1a9d4c3e" +down_revision = "195a1fb1d56e" +branch_labels = None +depends_on = None + + +def upgrade() -> None: + conn = op.get_bind() + inspector = inspect(conn) + columns = [col["name"] for col in inspector.get_columns("submissions")] + + if "ref_new_count" not in columns: + op.add_column( + "submissions", + sa.Column("ref_new_count", sa.Integer(), nullable=True, server_default="0"), + ) + + if "ref_reused_count" not in columns: + op.add_column( + "submissions", + sa.Column("ref_reused_count", sa.Integer(), nullable=True, server_default="0"), + ) + + +def downgrade() -> None: + conn = op.get_bind() + inspector = inspect(conn) + columns = [col["name"] for col in inspector.get_columns("submissions")] + + if "ref_reused_count" in columns: + op.drop_column("submissions", "ref_reused_count") + + if "ref_new_count" in columns: + op.drop_column("submissions", "ref_new_count") diff --git a/backend/alembic/versions/f1a2b3c4d5e6_add_image_and_infobox_counts_to_submissions.py b/backend/alembic/versions/f1a2b3c4d5e6_add_image_and_infobox_counts_to_submissions.py new file mode 100644 index 00000000..9bdfdf63 --- /dev/null +++ b/backend/alembic/versions/f1a2b3c4d5e6_add_image_and_infobox_counts_to_submissions.py @@ -0,0 +1,45 @@ +"""Add image_count and infobox_count columns to submissions table + +Revision ID: f1a2b3c4d5e6 +Revises: b3f7a9c2d1e4 + +""" +from alembic import op +import sqlalchemy as sa +from sqlalchemy import inspect + + +revision = "f1a2b3c4d5e6" +down_revision = "b3f7a9c2d1e4" +branch_labels = None +depends_on = None + + +def upgrade() -> None: + conn = op.get_bind() + inspector = inspect(conn) + columns = [col["name"] for col in inspector.get_columns("submissions")] + + if "image_count" not in columns: + op.add_column( + "submissions", + sa.Column("image_count", sa.Integer(), nullable=True), + ) + + if "infobox_count" not in columns: + op.add_column( + "submissions", + sa.Column("infobox_count", sa.Integer(), nullable=True), + ) + + +def downgrade() -> None: + conn = op.get_bind() + inspector = inspect(conn) + columns = [col["name"] for col in inspector.get_columns("submissions")] + + if "infobox_count" in columns: + op.drop_column("submissions", "infobox_count") + + if "image_count" in columns: + op.drop_column("submissions", "image_count") diff --git a/backend/app/models/contest.py b/backend/app/models/contest.py index 87afec41..e53b18bc 100644 --- a/backend/app/models/contest.py +++ b/backend/app/models/contest.py @@ -71,6 +71,10 @@ class Contest(BaseModel, ContestMixin): # Multi-parameter scoring configuration (stored as JSON string) scoring_parameters = db.Column(db.Text, nullable=True) + # Automated scoring configuration (stored as JSON string) + # Contains eligibility criteria and evaluation parameters + automated_settings = db.Column(db.Text, nullable=True) + # Submission type restriction: 'new', 'expansion', or 'both' allowed_submission_type = db.Column(db.String(20), default="both", nullable=False) @@ -155,6 +159,9 @@ def __init__(self, name, project_name, created_by, **kwargs): # Set scoring configuration (handles validation internally) self.set_scoring_parameters(kwargs.get("scoring_parameters")) + # Set automated scoring configuration (handles validation internally) + self.set_automated_settings(kwargs.get("automated_settings")) + # Set article requirements self.min_byte_count = kwargs.get("min_byte_count", 0) self.min_reference_count = kwargs.get("min_reference_count", 0) @@ -563,13 +570,153 @@ def get_scoring_mode(self): Get the current scoring mode for this contest. Returns: - str: 'simple' or 'multi_parameter' + str: 'simple', 'multi_parameter', or 'automated' """ + # Check for automated scoring mode first + automated = self.get_automated_settings() + if automated and automated.get("enabled") is True: + return "automated" + # Then check for multi-parameter scoring params = self.get_scoring_parameters() if params and params.get("enabled") is True: return "multi_parameter" return "simple" + # ------------------------------------------------------------------------ + # AUTOMATED EVALUATION ENGINE + # ------------------------------------------------------------------------ + + def evaluate_automated_submission(self, submission_data): + """ + Evaluate a submission against automated scoring criteria. + + Checks eligibility first, then calculates score if eligible. + + Args: + submission_data: Dict containing submission metadata: + - article_word_count: Article size in bytes + - incoming_links: Number of incoming links + - outgoing_links: Number of outgoing links + - ref_new_count: Number of new references + - ref_reused_count: Number of reused references + - image_count: Number of images + - infobox_count: Number of infoboxes + + Returns: + tuple: (is_eligible: bool, final_score: float, reason: str, breakdown: dict or None) + """ + automated = self.get_automated_settings() + if not automated or not automated.get("enabled"): + return False, 0, "Automated scoring not enabled for this contest", None + + eligibility = automated.get("eligibility", {}) + evaluation = automated.get("evaluation", {}) + + # --- ELIGIBILITY CHECKS --- + reasons = [] + + # Check minimum byte count (article_word_count stores bytes) + # NOTE: min_bytes is read from the common contest field (self.min_byte_count) + # instead of from automated_settings.eligibility.min_bytes (which has no UI + # and always defaults to 0). This ensures crawled articles are validated + # against the same threshold that manual submissions use. + # See PR #198 Comment #13 for full context on this unification. + min_bytes = self.min_byte_count or 0 + actual_bytes = submission_data.get("article_word_count") or 0 + if min_bytes > 0 and actual_bytes < min_bytes: + reasons.append(f"Article size ({actual_bytes} bytes) below minimum ({min_bytes} bytes)") + + # Check minimum incoming links + min_incoming = eligibility.get("min_incoming_links", 0) + actual_incoming = submission_data.get("incoming_links") or 0 + if min_incoming > 0 and actual_incoming < min_incoming: + reasons.append(f"Incoming links ({actual_incoming}) below minimum ({min_incoming})") + + # Check minimum outgoing links + min_outgoing = eligibility.get("min_outgoing_links", 0) + actual_outgoing = submission_data.get("outgoing_links") or 0 + if min_outgoing > 0 and actual_outgoing < min_outgoing: + reasons.append(f"Outgoing links ({actual_outgoing}) below minimum ({min_outgoing})") + + # Check minimum references + # NOTE: min_references is read from the common contest field + # (self.min_reference_count) for the same reason as min_bytes above. + min_refs = self.min_reference_count or 0 + actual_refs = (submission_data.get("ref_new_count") or 0) + (submission_data.get("ref_reused_count") or 0) + if min_refs > 0 and actual_refs < min_refs: + reasons.append(f"Total references ({actual_refs}) below minimum ({min_refs})") + + # If any eligibility check failed, return rejected + if reasons: + return False, 0, "; ".join(reasons), None + + # --- SCORE CALCULATION --- + score = 0.0 + breakdown = {} + + # Points per accepted article (base points) + base_points = float(evaluation.get("points_per_accepted", 0)) + score += base_points + breakdown["base_points"] = base_points + + # Points per byte + points_per_byte = float(evaluation.get("points_per_byte", 0)) + bytes_points = round(actual_bytes * points_per_byte, 2) + score += bytes_points + breakdown["bytes_points"] = bytes_points + breakdown["bytes_count"] = actual_bytes + + # Points per incoming link + points_per_incoming = float(evaluation.get("points_per_incoming_link", 0)) + incoming_points = round(actual_incoming * points_per_incoming, 2) + score += incoming_points + breakdown["incoming_links_points"] = incoming_points + breakdown["incoming_links_count"] = actual_incoming + + # Points per outgoing link + points_per_outgoing = float(evaluation.get("points_per_outgoing_link", 0)) + outgoing_points = round(actual_outgoing * points_per_outgoing, 2) + score += outgoing_points + breakdown["outgoing_links_points"] = outgoing_points + breakdown["outgoing_links_count"] = actual_outgoing + + # Points per new reference + points_per_new_ref = float(evaluation.get("points_per_new_reference", 0)) + new_refs = submission_data.get("ref_new_count") or 0 + new_ref_points = round(new_refs * points_per_new_ref, 2) + score += new_ref_points + breakdown["new_references_points"] = new_ref_points + breakdown["new_references_count"] = new_refs + + # Points per reused reference + points_per_reused_ref = float(evaluation.get("points_per_reused_reference", 0)) + reused_refs = submission_data.get("ref_reused_count") or 0 + reused_ref_points = round(reused_refs * points_per_reused_ref, 2) + score += reused_ref_points + breakdown["reused_references_points"] = reused_ref_points + breakdown["reused_references_count"] = reused_refs + + # Points per infobox + points_per_infobox = float(evaluation.get("points_per_infobox", 0)) + infobox_count = submission_data.get("infobox_count") or 0 + infobox_points = round(infobox_count * points_per_infobox, 2) + score += infobox_points + breakdown["infobox_points"] = infobox_points + breakdown["infobox_count"] = infobox_count + + # Points per image + points_per_image = float(evaluation.get("points_per_image", 0)) + image_count = submission_data.get("image_count") or 0 + image_points = round(image_count * points_per_image, 2) + score += image_points + breakdown["image_points"] = image_points + breakdown["image_count"] = image_count + + # Round score to 2 decimal places + final_score = round(score, 2) + + return True, final_score, f"Eligible. Score: {final_score}", breakdown + # ------------------------------------------------------------------------ # SERIALIZATION # ------------------------------------------------------------------------ @@ -607,6 +754,8 @@ def to_dict(self): ), # CRITICAL FIX: Explicitly add scoring_parameters "scoring_parameters": scoring_params, + # Automated scoring settings + "automated_settings": self.get_automated_settings(), # Computed fields "submission_count": self.get_submission_count(), "status": self.get_status(), diff --git a/backend/app/models/contest_mixin.py b/backend/app/models/contest_mixin.py index 0adb606c..67417432 100644 --- a/backend/app/models/contest_mixin.py +++ b/backend/app/models/contest_mixin.py @@ -168,6 +168,60 @@ def get_scoring_parameters(self): return None + # ------------------------------------------------------------------------ + # AUTOMATED SETTINGS MANAGEMENT (JSON Dictionary Storage) + # ------------------------------------------------------------------------ + + def set_automated_settings(self, settings): + """ + Set automated scoring settings configuration + + Args: + settings: Dict or None containing automated scoring configuration + { + "enabled": true, + "eligibility": { + "min_edits": 100, + "min_outgoing_links": 3 + }, + "evaluation": { + "points_per_accepted": 10, + "points_per_byte": 0.001, + "points_per_incoming_link": 2, + "points_per_outgoing_link": 1, + "points_per_category": 1, + "points_per_new_reference": 3, + "points_per_reused_reference": 1, + "points_per_infobox": 5, + "points_per_image": 2 + } + } + """ + if settings is None: + self.automated_settings = None + elif isinstance(settings, dict): + # Store as JSON string + self.automated_settings = json.dumps(settings) + else: + self.automated_settings = None + + + def get_automated_settings(self): + """ + Get automated scoring settings configuration + + Returns: + dict or None: Automated scoring settings configuration + """ + if not self.automated_settings: + return None + try: + # Parse JSON string back to dictionary + return json.loads(self.automated_settings) + except json.JSONDecodeError: + return None + + # ------------------------------------------------------------------------ # ORGANIZERS MANAGEMENT (Comma-Separated Storage) # ------------------------------------------------------------------------ diff --git a/backend/app/models/submission.py b/backend/app/models/submission.py index c520fd21..905a3665 100644 --- a/backend/app/models/submission.py +++ b/backend/app/models/submission.py @@ -70,6 +70,23 @@ class Submission(BaseModel): # Can be negative if article was reduced in size article_expansion_bytes = db.Column(db.Integer, nullable=True) + # Image count + image_count = db.Column(db.Integer, nullable=True) + + # Infobox count + infobox_count = db.Column(db.Integer, nullable=True) + + # Reference Analysis Metrics + ref_new_count = db.Column(db.Integer, nullable=True, default=0) + ref_reused_count = db.Column(db.Integer, nullable=True, default=0) + + # Link counts for future scoring evaluation + # Number of other mainspace articles that link to this article + incoming_links = db.Column(db.Integer, nullable=True) + + # Number of mainspace articles this article links to + outgoing_links = db.Column(db.Integer, nullable=True) + # Template enforcement tracking # True if template was automatically added to the article during submission template_added = db.Column(db.Boolean, nullable=True, default=False) @@ -91,6 +108,14 @@ class Submission(BaseModel): # Example: {"Quality": 8, "Sources": 7, "Neutrality": 9, "Formatting": 6} parameter_scores = db.Column(db.Text, nullable=True) + # Automated evaluation details (for automated scoring contests) + # Reason for rejection (if status is rejected) or success message + evaluation_reason = db.Column(db.Text, nullable=True) + + # Score breakdown as JSON (for accepted submissions in automated contests) + # Example: {"base_points": 10, "bytes_points": 5.2, "links_points": 3, ...} + score_breakdown = db.Column(db.Text, nullable=True) + # ------------------------------------------------------------------------ # Database Columns - Review Metadata @@ -173,6 +198,12 @@ def __init__( template_added=False, categories_added=None, category_error=None, + image_count=None, + infobox_count=None, + ref_new_count=0, + ref_reused_count=0, + incoming_links=None, + outgoing_links=None, ): """ Initialize a new Submission instance @@ -192,6 +223,12 @@ def __init__( template_added: Whether template was automatically added to article (optional) categories_added: List of category names that were automatically added (optional, stored as JSON) category_error: Error message if category attachment failed (optional) + image_count: Number of images in the article (optional) + infobox_count: Number of infoboxes in the article (optional) + ref_new_count: Number of new references added to the article (optional) + ref_reused_count: Number of reused references in the article (optional) + incoming_links: Number of other mainspace articles that link to this article (optional) + outgoing_links: Number of mainspace articles this article links to (optional) """ # Set required fields self.user_id = user_id @@ -219,6 +256,12 @@ def __init__( else: self.categories_added = None self.category_error = category_error + self.image_count = image_count + self.infobox_count = infobox_count + self.ref_new_count = ref_new_count + self.ref_reused_count = ref_reused_count + self.incoming_links = incoming_links + self.outgoing_links = outgoing_links self.reviewed_by = None self.reviewed_at = None self.review_comment = None @@ -316,6 +359,45 @@ def get_parameter_scores(self): except json.JSONDecodeError: return None + def get_score_breakdown(self): + """ + Get score breakdown for automated scoring + + Returns: + dict or None: Score breakdown with points per category + """ + if not self.score_breakdown: + return None + try: + return json.loads(self.score_breakdown) + except json.JSONDecodeError: + return None + + + # ------------------------------------------------------------------------ + # BYTE COUNT ALIAS (PR #198 Comment #9) + # ------------------------------------------------------------------------ + + @property + def article_byte_count(self): + """ + Clearer alias for the article_word_count column. + + HISTORICAL NOTE: The column is named 'article_word_count' but it stores + the article's size in BYTES as returned by the MediaWiki API 'size' field — + not a word count. The column name is intentionally left unchanged in the + database to avoid a risky Alembic migration. All NEW code should use this + alias instead of referencing article_word_count directly. + + See PR #198 Comment #9 for full context. + """ + return self.article_word_count + + @article_byte_count.setter + def article_byte_count(self, value): + """Set the article byte count (stored in the article_word_count column).""" + self.article_word_count = value + # ------------------------------------------------------------------------ # SUBMISSION STATUS UPDATE @@ -530,6 +612,12 @@ def to_dict(self, include_user_info=False): "article_page_id": self.article_page_id, "article_size_at_start": self.article_size_at_start, "article_expansion_bytes": self.article_expansion_bytes, + "image_count": self.image_count, + "infobox_count": self.infobox_count, + "ref_new_count": self.ref_new_count, + "ref_reused_count": self.ref_reused_count, + "incoming_links": self.incoming_links, + "outgoing_links": self.outgoing_links, "template_added": self.template_added, "categories_added": self.get_categories_added(), "category_error": self.category_error, @@ -542,6 +630,10 @@ def to_dict(self, include_user_info=False): # Multi-parameter scoring data "parameter_scores": self.get_parameter_scores(), + + # Automated evaluation details + "evaluation_reason": self.evaluation_reason, + "score_breakdown": self.get_score_breakdown(), } # Optionally include related user and contest information diff --git a/backend/app/routes/contest_routes.py b/backend/app/routes/contest_routes.py index 77e580d5..e5f973b0 100644 --- a/backend/app/routes/contest_routes.py +++ b/backend/app/routes/contest_routes.py @@ -4,8 +4,20 @@ """ from datetime import datetime, timezone +import os import traceback +# --------------------------------------------------------------------------- +# Category crawler rate-limiting constants (PR #198 Comment #6) +# --------------------------------------------------------------------------- +# Maximum articles that can ever be imported in a single crawl request. +# Keeps the value configurable on the server without touching code: +# export MAX_CRAWL_LIMIT=3000 +_MAX_CRAWL_LIMIT: int = int(os.environ.get("MAX_CRAWL_LIMIT", "2000")) + +# Default when the caller omits the `limit` field in the request body. +_DEFAULT_CRAWL_LIMIT: int = 500 + from flask import Blueprint, request, jsonify, current_app from sqlalchemy.exc import IntegrityError @@ -30,6 +42,12 @@ check_article_has_category, append_categories_to_article, get_article_reference_count, + get_detailed_reference_counts, + get_article_image_count, + get_article_infobox_count, + get_article_incoming_links, + get_article_outgoing_links, + crawl_category_articles, MEDIAWIKI_API_TIMEOUT, ) from app.services.outreach_dashboard import ( @@ -141,26 +159,26 @@ def get_all_contests(): def get_contest_outreach_data(contest_id): """ Get Outreach Dashboard course data for a contest - + Requires authentication - users must be logged in to view contest details. - + Args: contest_id: Contest ID - + Returns: JSON response with Outreach Dashboard course data or error message """ contest = Contest.query.get(contest_id) - + if not contest: return jsonify({"error": "Contest not found"}), 404 - + if not contest.outreach_dashboard_url: return jsonify({"error": "Contest does not have an Outreach Dashboard URL"}), 400 - + # Fetch course data from Outreach Dashboard API result = fetch_course_data(contest.outreach_dashboard_url) - + if result["success"]: return jsonify({ "success": True, @@ -179,24 +197,24 @@ def get_contest_outreach_data(contest_id): def get_outreach_dashboard_users(contest_id): """ Fetch Outreach Dashboard course users data for a contest. - + Args: contest_id: ID of the contest - + Returns: JSON response with Outreach Dashboard course users data or error message """ contest = Contest.query.get(contest_id) - + if not contest: return jsonify({"error": "Contest not found"}), 404 - + if not contest.outreach_dashboard_url: return jsonify({"error": "Contest does not have an Outreach Dashboard URL"}), 400 - + # Fetch course users data from Outreach Dashboard API result = fetch_course_users(contest.outreach_dashboard_url) - + if result["success"]: return jsonify({ "success": True, @@ -215,24 +233,24 @@ def get_outreach_dashboard_users(contest_id): def get_outreach_dashboard_articles(contest_id): """ Fetch Outreach Dashboard course articles data for a contest. - + Args: contest_id: ID of the contest - + Returns: JSON response with Outreach Dashboard course articles data or error message """ contest = Contest.query.get(contest_id) - + if not contest: return jsonify({"error": "Contest not found"}), 404 - + if not contest.outreach_dashboard_url: return jsonify({"error": "Contest does not have an Outreach Dashboard URL"}), 400 - + # Fetch course articles data from Outreach Dashboard API result = fetch_course_articles(contest.outreach_dashboard_url) - + if result["success"]: return jsonify({ "success": True, @@ -251,24 +269,24 @@ def get_outreach_dashboard_articles(contest_id): def get_outreach_dashboard_uploads(contest_id): """ Fetch Outreach Dashboard course uploads data for a contest. - + Args: contest_id: ID of the contest - + Returns: JSON response with Outreach Dashboard course uploads data or error message """ contest = Contest.query.get(contest_id) - + if not contest: return jsonify({"error": "Contest not found"}), 404 - + if not contest.outreach_dashboard_url: return jsonify({"error": "Contest does not have an Outreach Dashboard URL"}), 400 - + # Fetch course uploads data from Outreach Dashboard API result = fetch_course_uploads(contest.outreach_dashboard_url) - + if result["success"]: return jsonify({ "success": True, @@ -737,6 +755,69 @@ def create_contest(): # If no scoring_parameters provided, set to None (will use simple scoring) scoring_parameters = None + # ----------------------------------------------------------------------- + # Validate Automated Settings (Optional) + # ----------------------------------------------------------------------- + + automated_settings = data.get("automated_settings") + + if automated_settings: + if not isinstance(automated_settings, dict): + return jsonify({"error": "Automated settings must be an object"}), 400 + + # Validate automated scoring structure if enabled + if automated_settings.get("enabled"): + current_app.logger.info(f"[AUTOMATED CREATE] Automated scoring enabled") + + # Validate eligibility section + eligibility = automated_settings.get("eligibility", {}) + if not isinstance(eligibility, dict): + eligibility = {} + + # Validate evaluation section + evaluation = automated_settings.get("evaluation", {}) + if not isinstance(evaluation, dict): + evaluation = {} + + # Validate numeric values in eligibility + # NOTE: min_bytes and min_references are intentionally NOT included here. + # The automated evaluation engine reads those from the common contest fields + # (min_byte_count, min_reference_count) set via the UI, which are shared + # across all three scoring modes. This avoids the redundancy of having + # two separate sets of fields for the same thresholds. + # See PR #198 Comment #13 for full context. + for field in ["min_edits", "min_outgoing_links"]: + value = eligibility.get(field) + if value is not None: + try: + eligibility[field] = int(value) + if eligibility[field] < 0: + return jsonify({"error": f"{field} must be non-negative"}), 400 + except (ValueError, TypeError): + return jsonify({"error": f"{field} must be a valid integer"}), 400 + + # Validate numeric values in evaluation + for field in ["points_per_accepted", "points_per_byte", "points_per_incoming_link", + "points_per_outgoing_link", "points_per_category", "points_per_new_reference", + "points_per_reused_reference", "points_per_infobox", "points_per_image"]: + value = evaluation.get(field) + if value is not None: + try: + evaluation[field] = float(value) + if evaluation[field] < 0: + return jsonify({"error": f"{field} must be non-negative"}), 400 + except (ValueError, TypeError): + return jsonify({"error": f"{field} must be a valid number"}), 400 + + # Update with validated values + automated_settings["eligibility"] = eligibility + automated_settings["evaluation"] = evaluation + + # When automated mode is enabled, scoring_parameters should be null + scoring_parameters = None + else: + automated_settings = None + # ----------------------------------------------------------------------- # Create Contest # ----------------------------------------------------------------------- @@ -805,6 +886,7 @@ def create_contest(): template_link=template_link, outreach_dashboard_url=outreach_dashboard_url, scoring_parameters=scoring_parameters, + automated_settings=automated_settings, organizers=additional_organizers, min_reference_count=min_reference_count, ) @@ -1194,6 +1276,62 @@ def update_contest(contest_id): except ValueError as ve: return jsonify({"error": str(ve)}), 400 + # --- Automated Settings (Automated Scoring Mode) --- + if "automated_settings" in data: + as_settings = data.get("automated_settings") + + # Accept explicit null to disable automated settings + if as_settings is None: + contest.set_automated_settings(None) + elif not isinstance(as_settings, dict): + return jsonify({"error": "automated_settings must be an object"}), 400 + else: + # Validate automated scoring structure if enabled + if as_settings.get("enabled"): + # Validate eligibility section + eligibility = as_settings.get("eligibility", {}) + if not isinstance(eligibility, dict): + eligibility = {} + + # Validate evaluation section + evaluation = as_settings.get("evaluation", {}) + if not isinstance(evaluation, dict): + evaluation = {} + + # Validate numeric values in eligibility + for field in ["min_edits", "min_outgoing_links"]: + value = eligibility.get(field) + if value is not None: + try: + eligibility[field] = int(value) + if eligibility[field] < 0: + return jsonify({"error": f"{field} must be non-negative"}), 400 + except (ValueError, TypeError): + return jsonify({"error": f"{field} must be a valid integer"}), 400 + + # Validate numeric values in evaluation + for field in ["points_per_accepted", "points_per_byte", "points_per_incoming_link", + "points_per_outgoing_link", "points_per_category", "points_per_new_reference", + "points_per_reused_reference", "points_per_infobox", "points_per_image"]: + value = evaluation.get(field) + if value is not None: + try: + evaluation[field] = float(value) + if evaluation[field] < 0: + return jsonify({"error": f"{field} must be non-negative"}), 400 + except (ValueError, TypeError): + return jsonify({"error": f"{field} must be a valid number"}), 400 + + # Update with validated values + as_settings["eligibility"] = eligibility + as_settings["evaluation"] = evaluation + + # When automated mode is enabled, disable multi-parameter scoring + contest.set_scoring_parameters(None) + + # Persist validated automated settings + contest.set_automated_settings(as_settings) + # --- Organizers --- if "organizers" in data: organizers_payload = data.get("organizers") @@ -1271,6 +1409,39 @@ def submit_to_contest(contest_id): # pylint: disable=too-many-return-statements if not (article_link.startswith("http://") or article_link.startswith("https://")): return jsonify({"error": "Article link must be a valid URL"}), 400 + # --- Domain Validation Against Contest's Wiki (Automated Scoring Only) --- + # For automated scoring contests, ensure the submitted article belongs to + # the same wiki as the contest's configured categories. This prevents + # cross-wiki submissions (e.g., submitting a French Wikipedia article to + # an English Wikipedia contest). + # This check is scoped to automated contests only — simple and + # multi-parameter scoring modes are left untouched. + try: + _is_automated_contest = contest.get_scoring_mode() == "automated" + except Exception: # pylint: disable=broad-exception-caught + _is_automated_contest = False + + if _is_automated_contest: + contest_categories = contest.get_categories() + if contest_categories: + from urllib.parse import urlparse as _urlparse + allowed_domains = set() + for cat_url in contest_categories: + parsed = _urlparse(cat_url) + if parsed.netloc: + allowed_domains.add(parsed.netloc.lower()) + + if allowed_domains: + article_domain = _urlparse(article_link).netloc.lower() + if article_domain not in allowed_domains: + return jsonify({ + "error": ( + f"The article URL belongs to '{article_domain}', but this " + f"contest only accepts articles from: " + f"{', '.join(sorted(allowed_domains))}" + ) + }), 400 + # --- Contest Status Checks --- if not contest.is_active(): if contest.is_upcoming(): @@ -1301,6 +1472,10 @@ def submit_to_contest(contest_id): # pylint: disable=too-many-return-statements article_size_at_start = None article_expansion_bytes = None article_reference_count = None + ref_new_count = 0 + ref_reused_count = 0 + image_count = None + infobox_count = None # --- Fetch Article Information from MediaWiki API --- # MediaWiki API fetching has deep nesting due to complex error handling @@ -1686,6 +1861,83 @@ def submit_to_contest(contest_id): # pylint: disable=too-many-return-statements pass article_reference_count = None + # --- Fetch Detailed Metrics (Automated Scoring Only) --- + # These extra API calls are only needed for automated scoring evaluation. + # Skipping them for regular contests avoids MediaWiki API rate-limiting + # which was causing CSRF token failures during template/category enforcement. + incoming_links = None + outgoing_links = None + + is_automated_contest = False + try: + is_automated_contest = contest.get_scoring_mode() == "automated" + except Exception: # pylint: disable=broad-exception-caught + pass + + if is_automated_contest: + # Fetch detailed reference counts + try: + ref_counts = get_detailed_reference_counts(article_link) + ref_new_count = int(ref_counts.get("new", 0) or 0) + ref_reused_count = int(ref_counts.get("reused", 0) or 0) + except Exception as ref_detail_error: # pylint: disable=broad-exception-caught + try: + current_app.logger.warning( + "Failed to fetch detailed reference counts: %s", str(ref_detail_error) + ) + except Exception: # pylint: disable=broad-exception-caught + pass + ref_new_count = 0 + ref_reused_count = 0 + + # Fetch image count + try: + image_count = get_article_image_count(article_link) + except Exception as img_error: # pylint: disable=broad-exception-caught + try: + current_app.logger.warning( + "Failed to fetch image count: %s", str(img_error) + ) + except Exception: # pylint: disable=broad-exception-caught + pass + image_count = None + + # Fetch infobox count + try: + infobox_count = get_article_infobox_count(article_link) + except Exception as ibx_error: # pylint: disable=broad-exception-caught + try: + current_app.logger.warning( + "Failed to fetch infobox count: %s", str(ibx_error) + ) + except Exception: # pylint: disable=broad-exception-caught + pass + infobox_count = None + + # Fetch incoming links count + try: + incoming_links = get_article_incoming_links(article_link) + except Exception as incoming_error: # pylint: disable=broad-exception-caught + try: + current_app.logger.warning( + "Failed to fetch incoming links count: %s", str(incoming_error) + ) + except Exception: # pylint: disable=broad-exception-caught + pass + incoming_links = None + + # Fetch outgoing links count + try: + outgoing_links = get_article_outgoing_links(article_link) + except Exception as outgoing_error: # pylint: disable=broad-exception-caught + try: + current_app.logger.warning( + "Failed to fetch outgoing links count: %s", str(outgoing_error) + ) + except Exception: # pylint: disable=broad-exception-caught + pass + outgoing_links = None + # --- Validate Article Requirements --- # Validate article byte count against contest requirements # This check happens after fetching article information from MediaWiki API @@ -2047,6 +2299,12 @@ def submit_to_contest(contest_id): # pylint: disable=too-many-return-statements template_added=template_added, categories_added=categories_added, category_error=category_error, + image_count=image_count, + infobox_count=infobox_count, + ref_new_count=ref_new_count, + ref_reused_count=ref_reused_count, + incoming_links=incoming_links, + outgoing_links=outgoing_links, ) submission.save() @@ -2360,19 +2618,19 @@ def remove_contest_organizer(contest_id, username): def create_contest_request(): """ Create a contest creation request (for non-privileged users) - + Regular users who are not superadmin or trusted members can submit requests to create contests. Superadmins can review and approve/reject these requests. - + Expected JSON data: Same as create_contest endpoint - + Returns: JSON response with success message and request ID """ user = request.current_user data = request.validated_data - + # ----------------------------------------------------------------------- # Check if user already has permission to create contests # ----------------------------------------------------------------------- @@ -2381,26 +2639,26 @@ def create_contest_request(): return jsonify({ 'error': 'You already have permission to create contests. Use the regular create contest endpoint.' }), 400 - + # ----------------------------------------------------------------------- # Validate Required Fields (same as create_contest) # ----------------------------------------------------------------------- name = data["name"].strip() project_name = data["project_name"].strip() jury_members = data["jury_members"] - + if not name: return jsonify({"error": "Contest name is required"}), 400 - + if not project_name: return jsonify({"error": "Project name is required"}), 400 - + if not isinstance(jury_members, list) or len(jury_members) == 0: return ( jsonify({"error": "Jury members must be a non-empty array of usernames"}), 400, ) - + # ----------------------------------------------------------------------- # Validate Jury Members Exist in Database # ----------------------------------------------------------------------- @@ -2409,7 +2667,7 @@ def create_contest_request(): missing_users = [ username for username in jury_members if username not in existing_usernames ] - + if missing_users: return ( jsonify( @@ -2419,7 +2677,7 @@ def create_contest_request(): ), 400, ) - + # ----------------------------------------------------------------------- # Parse Optional Fields (same as create_contest) # ----------------------------------------------------------------------- @@ -2428,43 +2686,43 @@ def create_contest_request(): description = None else: description = str(description_value).strip() or None - + # Parse and validate dates start_date = validate_date_string(data.get("start_date")) end_date = validate_date_string(data.get("end_date")) - + # Validate date logic (end must be after start) if start_date and end_date and start_date >= end_date: return jsonify({"error": "End date must be after start date"}), 400 - + # Parse rules rules = data.get("rules", {}) if not isinstance(rules, dict): rules = {} - + # Parse scoring settings marks_accepted = data.get("marks_setting_accepted", 0) marks_rejected = data.get("marks_setting_rejected", 0) allowed_submission_type = data.get("allowed_submission_type", "both") - + try: marks_accepted = int(marks_accepted) marks_rejected = int(marks_rejected) except (ValueError, TypeError): return jsonify({"error": "Marks settings must be valid integers"}), 400 - + # Parse article requirements min_byte_count = data.get("min_byte_count") if min_byte_count is None: return jsonify({"error": "Minimum byte count is required"}), 400 - + try: min_byte_count = int(min_byte_count) if min_byte_count < 0: return jsonify({"error": "Minimum byte count must be non-negative"}), 400 except (ValueError, TypeError): return jsonify({"error": "Minimum byte count must be a valid integer"}), 400 - + min_reference_count = data.get("min_reference_count", 0) try: min_reference_count = int(min_reference_count) @@ -2472,12 +2730,12 @@ def create_contest_request(): return jsonify({"error": "Minimum reference count must be non-negative"}), 400 except (ValueError, TypeError): return jsonify({"error": "Minimum reference count must be a valid integer"}), 400 - + # Validate categories categories = data.get("categories") if not categories or not isinstance(categories, list) or len(categories) == 0: return jsonify({"error": "At least one category URL is required"}), 400 - + for category_url in categories: if not isinstance(category_url, str) or not category_url.strip(): return ( @@ -2491,13 +2749,13 @@ def create_contest_request(): jsonify({"error": "All category URLs must be valid HTTP/HTTPS URLs"}), 400, ) - + # Validate scoring parameters (same validation as create_contest) scoring_parameters = data.get("scoring_parameters") if scoring_parameters: if not isinstance(scoring_parameters, dict): return jsonify({"error": "Scoring parameters must be an object"}), 400 - + if scoring_parameters.get("enabled"): if "parameters" not in scoring_parameters: return ( @@ -2506,14 +2764,14 @@ def create_contest_request(): ), 400, ) - + parameters = scoring_parameters["parameters"] if not isinstance(parameters, list) or len(parameters) == 0: return ( jsonify({"error": "At least one scoring parameter is required"}), 400, ) - + total_weight = 0 for param in parameters: if not isinstance(param, dict): @@ -2532,13 +2790,13 @@ def create_contest_request(): total_weight += weight except (ValueError, TypeError): return jsonify({"error": "Weight must be a valid integer"}), 400 - + if total_weight != 100: return ( jsonify({"error": f"Weights must sum to 100, got {total_weight}"}), 400, ) - + # Parse template_link template_link = data.get('template_link') if template_link: @@ -2551,7 +2809,7 @@ def create_contest_request(): }), 400 else: template_link = None - + # ----------------------------------------------------------------------- # Create Contest Request # ----------------------------------------------------------------------- @@ -2560,7 +2818,7 @@ def create_contest_request(): additional_organizers = data.get('organizers', []) if not isinstance(additional_organizers, list): additional_organizers = [] - + # Create contest request instance contest_request = ContestRequest( user_id=user.id, @@ -2581,10 +2839,10 @@ def create_contest_request(): organizers=additional_organizers, min_reference_count=min_reference_count, ) - + # Save to database contest_request.save() - + return ( jsonify( { @@ -2594,7 +2852,7 @@ def create_contest_request(): ), 201, ) - + except Exception: # pylint: disable=broad-exception-caught # Log error internally but don't expose details to client return jsonify({"error": "Failed to create contest request"}), 500 @@ -2606,7 +2864,7 @@ def create_contest_request(): def get_contest_requests(): """ Get all contest creation requests (superadmin only) - + Returns: JSON response with list of contest requests """ @@ -2614,7 +2872,7 @@ def get_contest_requests(): requests = ContestRequest.query.filter_by(status='pending').order_by( ContestRequest.created_at.desc() ).all() - + return jsonify({ 'requests': [req.to_dict() for req in requests] }), 200 @@ -2626,29 +2884,29 @@ def get_contest_requests(): def approve_contest_request(request_id): """ Approve a contest creation request and create the contest (superadmin only) - + Args: request_id: Contest request ID to approve - + Returns: JSON response with success message and created contest ID """ user = request.current_user contest_request = ContestRequest.query.get(request_id) - + if not contest_request: return jsonify({'error': 'Contest request not found'}), 404 - + if contest_request.status != 'pending': return jsonify({ 'error': f'Request has already been {contest_request.status}' }), 400 - + # Get the requester user to use as contest creator requester = User.query.get(contest_request.user_id) if not requester: return jsonify({'error': 'Requester user not found'}), 404 - + # Create the contest from the request data try: contest = Contest( @@ -2670,22 +2928,22 @@ def approve_contest_request(request_id): organizers=contest_request.get_organizers(), min_reference_count=contest_request.min_reference_count, ) - + # Save contest to database contest.save() - + # Update request status contest_request.status = 'approved' contest_request.reviewed_by = user.id contest_request.reviewed_at = datetime.utcnow() contest_request.save() - + return jsonify({ 'message': 'Contest request approved and contest created successfully', 'contestId': contest.id, 'requestId': contest_request.id }), 200 - + except Exception as e: # pylint: disable=broad-exception-caught # Log error for debugging current_app.logger.error(f'Error approving contest request: {str(e)}') @@ -2700,38 +2958,168 @@ def approve_contest_request(request_id): def reject_contest_request(request_id): """ Reject a contest creation request (superadmin only) - + Args: request_id: Contest request ID to reject - + Expected JSON data (optional): rejection_reason: Reason for rejection - + Returns: JSON response with success message """ user = request.current_user # Get JSON data if provided (rejection_reason is optional) data = request.get_json() or {} - + contest_request = ContestRequest.query.get(request_id) - + if not contest_request: return jsonify({'error': 'Contest request not found'}), 404 - + if contest_request.status != 'pending': return jsonify({ 'error': f'Request has already been {contest_request.status}' }), 400 - + # Update request status contest_request.status = 'rejected' contest_request.reviewed_by = user.id contest_request.reviewed_at = datetime.utcnow() contest_request.rejection_reason = data.get('rejection_reason', '') contest_request.save() - + return jsonify({ 'message': 'Contest request rejected successfully', 'requestId': contest_request.id }), 200 + + +# ------------------------------------------------------------------------ +# CATEGORY CRAWLER ROUTE +# ------------------------------------------------------------------------ + + +@contest_bp.route("//crawl-category", methods=["POST"]) +@require_auth +@handle_errors +@validate_json_data(["category_url"]) +def crawl_category_for_contest(contest_id): + """ + Crawl articles from a Wikipedia category and create pending submissions. + """ + try: + user = request.current_user + data = request.get_json() + + # Fetch contest + contest = Contest.query.get(contest_id) + if not contest: + return jsonify({"error": "Contest not found"}), 404 + + # Check if contest uses automated scoring + try: + scoring_mode = contest.get_scoring_mode() + except Exception as e: + current_app.logger.error(f"Error getting scoring mode: {str(e)}") + scoring_mode = "simple" + + if scoring_mode != "automated": + return jsonify({ + "error": f"Category crawling is only available for automated scoring contests. Current mode: {scoring_mode}" + }), 400 + + # Permission check: jury member or superadmin only + jury_members = contest.get_jury_members() if hasattr(contest, "get_jury_members") else [] + is_jury_member = user.username in jury_members if jury_members else False + is_superadmin = getattr(user, "role", None) == "superadmin" + + if not (is_jury_member or is_superadmin): + return jsonify({"error": "You do not have permission to crawl categories for this contest"}), 403 + + # Get parameters + category_url = data.get("category_url") + + # Enforce rate limiting: cap imports per request to prevent server + # overload and MediaWiki API timeouts (PR #198 Comment #6). + # Default: 500 | Hard cap: MAX_CRAWL_LIMIT (default 2000, env-configurable) + try: + requested_limit = int(data.get("limit", _DEFAULT_CRAWL_LIMIT)) + limit = min(max(requested_limit, 1), _MAX_CRAWL_LIMIT) + except (ValueError, TypeError): + limit = _DEFAULT_CRAWL_LIMIT + + # Optional cmcontinue token from a previous crawl batch. + # When present the crawler resumes from this position in the category + # instead of starting from the beginning (supports "Import Next Batch"). + continue_from = data.get("continue_from") or None + + # Crawl the category + result = crawl_category_articles( + category_url, + limit=limit, + continue_from=continue_from, + ) + + if not result: + return jsonify({ + "error": "Failed to crawl category. Please check the category URL." + }), 400 + + articles = result.get("articles", []) + imported = [] + skipped = 0 + + # Get existing article links for this contest to avoid duplicates + existing_links = set( + s.article_link for s in Submission.query.filter_by(contest_id=contest_id).all() + ) + + # Create submissions for each article + for article in articles: + article_url = article.get("url") + article_title = article.get("title") + + # Skip if already submitted + if article_url in existing_links: + skipped += 1 + continue + + # Create pending submission + submission = Submission( + user_id=user.id, + contest_id=contest_id, + article_title=article_title, + article_link=article_url, + status="pending", + ) + + try: + submission.save() + imported.append(article_title) + existing_links.add(article_url) + except IntegrityError: + db.session.rollback() + skipped += 1 + except Exception as e: + current_app.logger.error(f"Error creating submission for {article_title}: {str(e)}") + db.session.rollback() + skipped += 1 + + return jsonify({ + "message": f"Successfully imported {len(imported)} articles from category", + "total_imported": len(imported), + "skipped": skipped, + "category": result.get("category"), + "wiki_base": result.get("wiki_base"), + "articles": imported[:100], + # Pagination: pass next_continue back to the client so it can + # request the next batch with "Import Next Batch" (Option B). + "has_more": result.get("has_more", False), + "next_continue": result.get("next_continue"), + }), 200 + + + except Exception as e: + current_app.logger.error(f"crawl_category_for_contest error: {str(e)}\n{traceback.format_exc()}") + return jsonify({"error": f"Internal server error: {str(e)}"}), 500 diff --git a/backend/app/routes/submission_routes.py b/backend/app/routes/submission_routes.py index 20f70761..13e56dd1 100644 --- a/backend/app/routes/submission_routes.py +++ b/backend/app/routes/submission_routes.py @@ -3,6 +3,7 @@ Handles submission management and review functionality """ +import json from urllib.parse import urlparse from datetime import datetime @@ -25,7 +26,8 @@ get_latest_revision_author, build_mediawiki_revisions_api_params, get_mediawiki_headers, - MEDIAWIKI_API_TIMEOUT + get_detailed_reference_counts, + MEDIAWIKI_API_TIMEOUT, ) # Create blueprint @@ -249,29 +251,94 @@ def get_submission_stats(): @handle_errors def refresh_metadata(contest_id): """ - Refresh article metadata (word count, author, etc.) for all submissions in a contest. + Refresh article metadata for submissions in a contest. - This endpoint fetches the latest metadata from MediaWiki API for all submissions - in the specified contest and updates the database with the current values. + Supports offset-based pagination so large contests can be processed in + multiple smaller batches without hitting server or MediaWiki API timeouts. - Args: - contest_id: The ID of the contest to refresh submissions for + Query parameters: + offset (int, default 0) — how many submissions to skip before + starting this batch. + batch_size (int, default 50) — how many submissions to process in this + call. Capped at 100 to prevent timeouts. - Returns: - JSON response with refresh results + Response includes pagination fields: + has_more — True if there are more submissions after this batch. + next_offset — Pass this as `offset` in the next request. + total_count — Total number of submissions in the contest. + + PR #198 Comment #11. """ user = request.current_user # Validate contest access and permissions - # Note: contest variable is validated but not used in this route - _contest, error_response = validate_contest_submission_access( + contest, error_response = validate_contest_submission_access( contest_id, user, Contest ) if error_response: return error_response - # Get all submissions for this contest - submissions = Submission.query.filter_by(contest_id=contest_id).all() + # Check if this is an automated scoring contest + is_automated = False + try: + is_automated = contest.get_scoring_mode() == "automated" + scoring_mode = contest.get_scoring_mode() + current_app.logger.info( + "Contest %s scoring mode: %s, is_automated: %s", + contest_id, scoring_mode, is_automated + ) + except Exception as scoring_err: # pylint: disable=broad-exception-caught + current_app.logger.warning( + "Failed to check scoring mode: %s", str(scoring_err) + ) + + # ------------------------------------------------------------------ # + # Pagination — automated scoring only # + # For simple / multi-parameter contests we preserve the original # + # behaviour: process every submission in a single request. # + # ------------------------------------------------------------------ # + if is_automated: + # IMPORTANT: Each article requires ~6 Wikipedia API calls in + # automated mode (revisions, refs, incoming/outgoing links, + # images, infoboxes). A batch of 50 = ~300 network calls which + # can easily take 10+ minutes and appear "stuck". + # Default is 10 (≈60 calls, completes in under 60 s). + _DEFAULT_BATCH_SIZE = 10 + _MAX_BATCH_SIZE = 50 + + try: + offset = max(0, int(request.args.get("offset", 0))) + except (ValueError, TypeError): + offset = 0 + + try: + batch_size = min( + max(1, int(request.args.get("batch_size", _DEFAULT_BATCH_SIZE))), + _MAX_BATCH_SIZE, + ) + except (ValueError, TypeError): + batch_size = _DEFAULT_BATCH_SIZE + + total_count = Submission.query.filter_by(contest_id=contest_id).count() + + submissions = ( + Submission.query + .filter_by(contest_id=contest_id) + .order_by(Submission.id) + .offset(offset) + .limit(batch_size) + .all() + ) + else: + # Non-automated: fetch ALL submissions at once (original behaviour) + offset = 0 + submissions = ( + Submission.query + .filter_by(contest_id=contest_id) + .order_by(Submission.id) + .all() + ) + total_count = len(submissions) if not submissions: return ( @@ -280,7 +347,10 @@ def refresh_metadata(contest_id): "message": "No submissions found for this contest", "updated": 0, "failed": 0, - "total": 0, + "total": total_count, + "has_more": False, + "next_offset": offset, + "total_count": total_count, } ), 200, @@ -409,10 +479,9 @@ def calculate_expansion_bytes(submission_item, article_info): except Exception as exp_error: # pylint: disable=broad-exception-caught # If expansion calculation fails, log but don't fail the update try: - from flask import current_app - current_app.logger.warning( - f"Failed to calculate expansion for submission {submission_item.id}: {str(exp_error)}" + "Failed to calculate expansion for submission %s: %s", + submission_item.id, str(exp_error) ) except Exception: # pylint: disable=broad-exception-caught pass @@ -447,8 +516,12 @@ def calculate_expansion_bytes(submission_item, article_info): submission.article_created_at = timestamp_str else: submission.article_created_at = None - # Do NOT update article_word_count - it should remain fixed at submission time - # article_word_count represents the size at the time of submission + + # For crawler-imported submissions (no article_word_count), fetch it from API + # This is needed for automated evaluation + if not submission.article_word_count and info.get("current_size"): + submission.article_word_count = info["current_size"] + if info.get("article_page_id"): submission.article_page_id = info["article_page_id"] @@ -460,6 +533,106 @@ def calculate_expansion_bytes(submission_item, article_info): # These should remain fixed at submission time calculate_expansion_bytes(submission, info) + # Update detailed reference counts on refresh + try: + ref_counts = get_detailed_reference_counts(submission.article_link) + submission.ref_new_count = int(ref_counts.get("new", 0) or 0) + submission.ref_reused_count = int(ref_counts.get("reused", 0) or 0) + except Exception as ref_error: # pylint: disable=broad-exception-caught + try: + current_app.logger.warning( + "Failed to refresh reference counts for submission %s: %s", + submission.id, + str(ref_error), + ) + except Exception: # pylint: disable=broad-exception-caught + pass + + # --- Automated Scoring Evaluation --- + # For automated contests, fetch additional metrics and evaluate + if is_automated: + current_app.logger.info( + "Evaluating submission %s in automated mode", + submission.id + ) + try: + # Fetch incoming/outgoing links + from app.utils import ( # pylint: disable=import-outside-toplevel + get_article_incoming_links, + get_article_outgoing_links, + get_article_image_count, + get_article_infobox_count, + ) + + incoming = get_article_incoming_links(submission.article_link) or 0 + outgoing = get_article_outgoing_links(submission.article_link) or 0 + images = get_article_image_count(submission.article_link) or 0 + infoboxes = get_article_infobox_count(submission.article_link) or 0 + + current_app.logger.info( + "Submission %s: incoming=%s, outgoing=%s, images=%s, infoboxes=%s", + submission.id, incoming, outgoing, images, infoboxes + ) + + # Update submission with link counts + submission.incoming_links = incoming + submission.outgoing_links = outgoing + submission.image_count = images + submission.infobox_count = infoboxes + + # Evaluate submission against automated criteria + submission_data = { + "article_word_count": submission.article_word_count, + "incoming_links": incoming, + "outgoing_links": outgoing, + "ref_new_count": submission.ref_new_count, + "ref_reused_count": submission.ref_reused_count, + "image_count": images, + "infobox_count": infoboxes, + } + + current_app.logger.info( + "Submission %s data for evaluation: %s", + submission.id, submission_data + ) + + is_eligible, final_score, reason, breakdown = ( + contest.evaluate_automated_submission(submission_data) + ) + + current_app.logger.info( + "Submission %s result: eligible=%s, score=%s, reason=%s", + submission.id, is_eligible, final_score, reason + ) + + # Update submission status, score, and evaluation details + if is_eligible: + submission.status = "accepted" + submission.score = int(round(final_score)) + submission.evaluation_reason = reason + submission.score_breakdown = json.dumps(breakdown) if breakdown else None + else: + submission.status = "rejected" + submission.score = 0 + submission.evaluation_reason = reason + submission.score_breakdown = None + + current_app.logger.info( + "Automated evaluation for submission %s: %s - %s", + submission.id, + submission.status, + reason, + ) + + except Exception as eval_error: # pylint: disable=broad-exception-caught + current_app.logger.error( + "Failed to evaluate submission %s: %s", + submission.id, + str(eval_error), + ) + import traceback # pylint: disable=import-outside-toplevel + current_app.logger.error(traceback.format_exc()) + updated += 1 else: failed += 1 @@ -471,28 +644,116 @@ def calculate_expansion_bytes(submission_item, article_info): db.session.rollback() return jsonify({"error": "Failed to save updates to database"}), 500 + # ------------------------------------------------------------------ # + # Build response # + # ------------------------------------------------------------------ # + if is_automated: + # Automated: paginated response so frontend can loop + processed_up_to = offset + len(submissions) + has_more = processed_up_to < total_count + next_offset = processed_up_to if has_more else offset + + return ( + jsonify( + { + "message": f"Refreshed metadata for {updated} submissions", + "updated": updated, + "failed": failed, + "total": total_count, + # Pagination fields (PR #198 Comment #11) + "has_more": has_more, + "next_offset": next_offset, + "total_count": total_count, + "batch_size": batch_size, + "offset": offset, + } + ), + 200, + ) + + # Non-automated: simple response (original behaviour, no pagination) return ( jsonify( { "message": f"Refreshed metadata for {updated} submissions", "updated": updated, "failed": failed, - "total": len(submissions), + "total": total_count, } ), 200, ) +# ------------------------------------------------------------------------ +# SUBMISSION DELETION ENDPOINT +# ------------------------------------------------------------------------ + +@submission_bp.route("/", methods=["DELETE"]) +@require_auth +@handle_errors +def delete_submission(submission_id): + """ + Delete a specific submission by ID. + + Only admins, jury members of the contest, and contest creators/organizers + are allowed to delete submissions. When a submission is deleted its score + is subtracted from the submitter's total so the leaderboard stays accurate. + + Args: + submission_id: Submission ID + + Returns: + JSON response confirming deletion + """ + user = request.current_user + + # Load submission with related data for permission check and score update + submission = Submission.query.options( + joinedload(Submission.submitter), + joinedload(Submission.contest), + ).get(submission_id) + + if not submission: + return jsonify({"error": "Submission not found"}), 404 + + # Permission check using existing model method + if not submission.can_be_deleted_by(user): + return jsonify({"error": "You are not allowed to delete this submission"}), 403 + + try: + # Subtract the submission's score from the user's total before deleting + if submission.score and submission.submitter: + submission.submitter.update_score(-submission.score) + + # Remove the submission from the database + submission.delete() + + return jsonify({"message": "Submission deleted successfully"}), 200 + + except Exception as delete_err: # pylint: disable=broad-exception-caught + db.session.rollback() + current_app.logger.error( + "Failed to delete submission %s: %s", submission_id, str(delete_err) + ) + return jsonify({"error": "Failed to delete submission"}), 500 + + # ------------------------------------------------------------------------ # SUBMISSION REVIEW ENDPOINT # ------------------------------------------------------------------------ + + + + + + @submission_bp.route("//review", methods=["PUT"]) @require_auth @handle_errors @validate_json_data(["status"]) -def review_submission(submission_id): +def review_submission(submission_id): # pylint: disable=too-many-return-statements user = request.current_user data = request.validated_data @@ -574,9 +835,9 @@ def review_submission(submission_id): contest=contest, parameter_scores=parameter_scores, ) - except Exception as e: # pylint: disable=broad-exception-caught + except Exception as review_err: # pylint: disable=broad-exception-caught db.session.rollback() - print(f"Review error (multi-parameter): {str(e)}") + print(f"Review error (multi-parameter): {str(review_err)}") return jsonify({"error": "Internal server error"}), 500 else: @@ -591,9 +852,9 @@ def review_submission(submission_id): comment=comment, contest=contest, ) - except Exception as e: # pylint: disable=broad-exception-caught + except Exception as review_err: # pylint: disable=broad-exception-caught db.session.rollback() - print(f"Review error (rejected, multi): {str(e)}") + print(f"Review error (rejected, multi): {str(review_err)}") return jsonify({"error": "Internal server error"}), 500 # --- Simple Scoring Mode --- @@ -628,9 +889,9 @@ def review_submission(submission_id): comment=comment, contest=contest, ) - except Exception as e: # pylint: disable=broad-exception-caught + except Exception as review_err: # pylint: disable=broad-exception-caught db.session.rollback() - print(f"Review error (simple): {str(e)}") + print(f"Review error (simple): {str(review_err)}") return jsonify({"error": "Internal server error"}), 500 return ( diff --git a/backend/app/routes/user_routes.py b/backend/app/routes/user_routes.py index bf4d2138..c99dd58d 100644 --- a/backend/app/routes/user_routes.py +++ b/backend/app/routes/user_routes.py @@ -254,15 +254,14 @@ def get_dashboard(): JSON response with user's dashboard information """ user = request.current_user - # Get user's total score total_score = user.score + # --- Get Contest-wise Scores --- - # --- Get Contest-wise Scores --- from app.models.submission import Submission from app.models.contest import Contest - # Query submissions grouped by contest to calculate scores + # Contest-wise scores contest_scores = db.session.query( Contest.id.label('contest_id'), Contest.name.label('contest_name'), @@ -272,7 +271,7 @@ def get_dashboard(): Submission.user_id == user.id ).group_by(Contest.id, Contest.name).order_by(Contest.name).all() - # --- Get All User's Submissions Grouped by Contest --- + # Submissions grouped by contest submissions_query = db.session.query( Submission, Contest.name.label('contest_name') @@ -280,7 +279,6 @@ def get_dashboard(): Submission.user_id == user.id ).order_by(Submission.submitted_at.desc()).all() - # Group submissions by contest for organized display submissions_by_contest = {} for submission, contest_name in submissions_query: contest_id = submission.contest_id @@ -292,7 +290,7 @@ def get_dashboard(): } submissions_by_contest[contest_id]['submissions'].append(submission.to_dict()) - # --- Get Contests Created by User --- + # Created contests created_contests = Contest.query.filter_by(created_by=user.username).all() created_contests_data = [] for contest in created_contests: @@ -300,7 +298,7 @@ def get_dashboard(): contest_data['submission_count'] = contest.get_submission_count() created_contests_data.append(contest_data) - # --- Get Contests Where User is a Jury Member --- + # Jury contests jury_contests = Contest.query.filter( Contest.jury_members.like(f'%{user.username}%') ).all() @@ -310,6 +308,24 @@ def get_dashboard(): contest_data['submission_count'] = contest.get_submission_count() jury_contests_data.append(contest_data) + # Participated contests (contests jisme user ne submit kiya) + participated_contests_query = db.session.query( + Contest, + db.func.min(Submission.submitted_at).label('submitted_at') + ).join( + Submission, Contest.id == Submission.contest_id + ).filter( + Submission.user_id == user.id + ).group_by(Contest.id).order_by( + db.func.min(Submission.submitted_at).desc() + ).all() + + participated_contests_data = [] + for contest, submitted_at in participated_contests_query: + contest_data = contest.to_dict() + contest_data['submitted_at'] = submitted_at.isoformat() if submitted_at else None + participated_contests_data.append(contest_data) + return jsonify({ 'username': user.username, 'total_score': total_score, @@ -324,10 +340,10 @@ def get_dashboard(): ], 'submissions_by_contest': list(submissions_by_contest.values()), 'created_contests': created_contests_data, - 'jury_contests': jury_contests_data + 'jury_contests': jury_contests_data, + 'participated_contests': participated_contests_data # NEW }), 200 - @user_bp.route('/all', methods=['GET']) @require_role('admin') @handle_errors @@ -816,11 +832,11 @@ def request_trusted_member(): Request trusted member status (creator account request) This endpoint handles creator account requests for users who logged in via MediaWiki OAuth. - + Workflow: 1. If user has >= 300 edits: automatically grant trusted member status 2. If user has < 300 edits: require a reason and submit for superadmin review - + Only users who logged in via MediaWiki OAuth can request creator accounts. Superadmins don't need permission - they can create contests directly. @@ -1135,4 +1151,4 @@ def remove_trusted_member(user_id): return jsonify({ 'message': f'User {user.username} has been removed from trusted members' - }), 200 \ No newline at end of file + }), 200 diff --git a/backend/app/utils/__init__.py b/backend/app/utils/__init__.py index e67a9de4..15eb3012 100644 --- a/backend/app/utils/__init__.py +++ b/backend/app/utils/__init__.py @@ -43,7 +43,13 @@ "check_article_has_category", "append_categories_to_article", "get_article_reference_count", + "get_detailed_reference_counts", "get_mediawiki_user_edit_count", + "get_article_image_count", + "get_article_infobox_count", + "get_article_incoming_links", + "get_article_outgoing_links", + "crawl_category_articles", ] @@ -159,6 +165,8 @@ def extract_page_title_from_url(article_url: str) -> Optional[str]: return None + + def build_mediawiki_revisions_api_params(page_title: str) -> Dict[str, Any]: # pylint: disable=invalid-name """ Build a standard parameter set for MediaWiki `revisions` queries. @@ -384,6 +392,9 @@ def get_article_wikitext(article_url: str) -> Optional[str]: # pylint: disable= Returns: Wikitext content as string, or None if fetch fails. """ + if not article_url: + return None + page_title = extract_page_title_from_url(article_url) if not page_title: return None @@ -440,6 +451,78 @@ def get_article_wikitext(article_url: str) -> Optional[str]: # pylint: disable= return content +def get_detailed_reference_counts(article_url: str) -> Dict[str, int]: + """ + Count new (defined) and reused (self-closing) tags in a wiki article. + + New refs: content — define a source inline for the first time. + Reused refs: — self-closing back-reference to a named source. + + Implementation uses mwparserfromhell, a purpose-built MediaWiki wikitext parser. + This correctly handles: + - ... blocks (refs inside are not counted) + - Adjacent refs: AB (counted as 2, not 1) + - Named refs appearing with full content more than once + - Malformed / nested markup that confuses regex-based approaches + + Falls back to a simple regex count if mwparserfromhell is unavailable at runtime, + so the server never crashes due to a missing dependency — it just counts less + accurately and logs a warning. + + PR #198 Comment #7. + """ + wikitext = get_article_wikitext(article_url) + if not wikitext: + return {"new": 0, "reused": 0} + + # ------------------------------------------------------------------ # + # Primary path: proper wikitext parser # + # ------------------------------------------------------------------ # + try: + import mwparserfromhell # pylint: disable=import-outside-toplevel + + parsed = mwparserfromhell.parse(wikitext) + + new_refs = 0 + reused_refs = 0 + + for tag in parsed.filter_tags(): + # mwparserfromhell normalises tag names; compare lower-case + if str(tag.tag).strip().lower() != "ref": + continue + if tag.self_closing: + # — back-reference to a named source + reused_refs += 1 + else: + # content — inline source definition + new_refs += 1 + + return {"new": new_refs, "reused": reused_refs} + + except ImportError: + # ------------------------------------------------------------------ # + # Fallback path: regex (less accurate, but never crashes the server) # + # ------------------------------------------------------------------ # + import logging # pylint: disable=import-outside-toplevel + logging.warning( + "mwparserfromhell is not installed — falling back to regex for " + "reference counting. Install it with: pip install mwparserfromhell" + ) + + # Strip HTML comments so commented-out refs aren't counted + text = re.sub(r"", "", wikitext, flags=re.DOTALL) + + reused_refs = len(re.findall(r"]*?/>", text, flags=re.IGNORECASE)) + new_refs = len(re.findall( + r"][^>]*>(?:[^<]|<(?!/ref>))*", + text, + flags=re.IGNORECASE | re.DOTALL, + )) + return {"new": new_refs, "reused": reused_refs} + + + + def check_article_has_template(article_url: str, template_name: str) -> Dict[str, Any]: """ Check if an article begins with the specified template. @@ -948,6 +1031,63 @@ def _fetch_footnotes_count(api_url: str, page_title: str, headers: dict) -> int: return 0 +def _log_warning(message: str, error: Exception) -> None: + """Best-effort logging helper that uses Flask current_app when available. + + This keeps network helpers free from hard Flask dependencies while still + providing useful diagnostics in a running application. + """ + try: + from flask import current_app + + current_app.logger.warning("%s: %s", message, str(error)) + except Exception: # pylint: disable=broad-exception-caught + # Logging must never break core logic, so ignore any logging failures + pass + + +def get_article_image_count(article_url: str) -> Optional[int]: + """ + The count is approximate and based purely on wikitext patterns; it does + not guarantee that every match results in a rendered image, but it + generally tracks user-added content images. + """ + try: + wikitext = get_article_wikitext(article_url) + if wikitext is None: + return None + + # Match explicit file/image links like [[File:Example.jpg|...]] or + # [[Image:Example.png|...]] in a case-insensitive way. + matches = re.findall(r'\[\[(?:File|Image):', wikitext, flags=re.IGNORECASE) + return len(matches) + + except Exception as error: # pylint: disable=broad-exception-caught + _log_warning("Failed to fetch image count", error) + return None + + +def get_article_infobox_count(article_url: str) -> Optional[int]: + """Count approximate number of infobox templates in article wikitext. + + Detection is done via a simple regex scan for ``{{infobox ...}}`` in the + raw wikitext. This is an approximation and may over-count or under-count + in edge cases (e.g. nested templates, unusual formatting), but is + sufficient for high-level richness metrics. + """ + try: + wikitext = get_article_wikitext(article_url) + if wikitext is None: + return None + + matches = re.findall(r"\{\{\s*infobox\b", wikitext, flags=re.IGNORECASE) + return len(matches) + + except Exception as error: # pylint: disable=broad-exception-caught + _log_warning("Failed to fetch infobox count", error) + return None + + def get_article_reference_count(article_url: str) -> Optional[int]: """ Get the total number of references in a MediaWiki article. @@ -1430,3 +1570,334 @@ def append_categories_to_article( # pylint: disable=too-many-return-statements return result result['error'] = 'Unknown API response format' return result + + +def get_article_incoming_links(article_url: str) -> Optional[int]: + """ + Count the number of mainspace articles that link to the given article. + + This uses the MediaWiki API's "backlinks" query to count incoming links + from other articles in the main namespace (namespace 0). + + Args: + article_url: Full URL to the wiki article + + Returns: + Integer count of incoming links from mainspace articles, or None if fetch fails + """ + try: + # Extract page title from URL + page_title = extract_page_title_from_url(article_url) + if not page_title: + return None + + # Parse the article URL to extract base URL + url_obj = urlparse(article_url) + base_url = f"{url_obj.scheme}://{url_obj.netloc}" + api_url = f"{base_url}/w/api.php" + + # Build API parameters for backlinks query + params = { + "action": "query", + "format": "json", + "formatversion": "2", + "list": "backlinks", + "bltitle": page_title, + "blnamespace": "0", # Only count links from mainspace (namespace 0) + "bllimit": "500", # Maximum allowed by API + "blfilterredir": "nonredirects", # Exclude redirects + "redirects": "true", # Follow redirects to get actual page + "converttitles": "true", + } + + headers = get_mediawiki_headers() + + # Make initial request + response = requests.get(api_url, params=params, headers=headers, timeout=MEDIAWIKI_API_TIMEOUT) + if response.status_code != 200: + return None + + data = response.json() + if "error" in data: + return None + + # Count backlinks from response + backlinks = data.get("query", {}).get("backlinks", []) + total_count = len(backlinks) + + # Check if there are more results (continuation) + continue_params = data.get("continue") + while continue_params and total_count < 10000: # Safety limit to prevent infinite loops + # Update params with continuation token + params.update(continue_params) + + response = requests.get(api_url, params=params, headers=headers, timeout=MEDIAWIKI_API_TIMEOUT) + if response.status_code != 200: + break + + data = response.json() + if "error" in data: + break + + # Add more backlinks to count + more_backlinks = data.get("query", {}).get("backlinks", []) + total_count += len(more_backlinks) + + # Get next continuation token + continue_params = data.get("continue") + + return total_count + + except Exception: # pylint: disable=broad-exception-caught + # If link counting fails, return None to indicate failure + # This ensures submission process continues even if link counting fails + return None + + +def get_article_outgoing_links(article_url: str) -> Optional[int]: + """ + Count the number of mainspace articles that the given article links to. + + This uses the MediaWiki API's "links" query to count outgoing links + to other articles in the main namespace (namespace 0). + + Args: + article_url: Full URL to the wiki article + + Returns: + Integer count of outgoing links to mainspace articles, or None if fetch fails + """ + try: + # Extract page title from URL + page_title = extract_page_title_from_url(article_url) + if not page_title: + return None + + # Parse the article URL to extract base URL + url_obj = urlparse(article_url) + base_url = f"{url_obj.scheme}://{url_obj.netloc}" + api_url = f"{base_url}/w/api.php" + + # Build API parameters for links query + params = { + "action": "query", + "format": "json", + "formatversion": "2", + "prop": "links", + "titles": page_title, + "plnamespace": "0", # Only count links to mainspace (namespace 0) + "pllimit": "500", # Maximum allowed by API + "plfilterredir": "nonredirects", # Exclude redirects + "redirects": "true", # Follow redirects to get actual page + "converttitles": "true", + } + + headers = get_mediawiki_headers() + + # Make initial request + response = requests.get(api_url, params=params, headers=headers, timeout=MEDIAWIKI_API_TIMEOUT) + if response.status_code != 200: + return None + + data = response.json() + if "error" in data: + return None + + # Extract links from response + pages = data.get("query", {}).get("pages", []) + if not pages: + return None + + page_data = pages[0] + if page_data.get("missing", False): + return None + + links = page_data.get("links", []) + total_count = len(links) + + # Check if there are more results (continuation) + continue_params = data.get("continue") + while continue_params and total_count < 10000: # Safety limit to prevent infinite loops + # Update params with continuation token + params.update(continue_params) + + response = requests.get(api_url, params=params, headers=headers, timeout=MEDIAWIKI_API_TIMEOUT) + if response.status_code != 200: + break + + data = response.json() + if "error" in data: + break + + # Add more links to count + pages = data.get("query", {}).get("pages", []) + if pages: + more_links = pages[0].get("links", []) + total_count += len(more_links) + + # Get next continuation token + continue_params = data.get("continue") + + return total_count + + except Exception: # pylint: disable=broad-exception-caught + # If link counting fails, return None to indicate failure + # This ensures submission process continues even if link counting fails + return None + + +# --------------------------------------------------------------------------- +# Category Crawler Utilities +# --------------------------------------------------------------------------- + +def crawl_category_articles( + category_url: str, + limit: int = 5000, + mw_uri: Optional[str] = None, + continue_from: Optional[str] = None, +) -> Optional[Dict[str, Any]]: + """ + Crawl articles from a Wikipedia category using the MediaWiki API. + + Uses the `list=categorymembers` API to fetch all pages in a category, + handling pagination via `cmcontinue` tokens. + + Args: + category_url: Full URL to a Wikipedia category page. + limit: Maximum number of articles to fetch in this call. + mw_uri: Optional MediaWiki API base URI. Extracted from + category_url when omitted. + continue_from: A ``cmcontinue`` token returned by a previous call. + When provided the crawl resumes from that position + instead of starting from the beginning of the category. + Pass the value of ``next_continue`` from the previous + response to implement "Import Next Batch" behaviour. + + Returns: + Dictionary with: + - "articles": List of dicts with "title", "url", "page_id". + - "total": Number of articles fetched in this call. + - "category": Category name extracted from URL. + - "wiki_base": Wiki base URL (e.g., "https://en.wikipedia.org"). + - "has_more": True if there are more articles beyond this batch. + - "next_continue": cmcontinue token to pass as ``continue_from`` + in the next call (None when has_more is False). + Or None if an error occurs. + """ + try: + # Enforce maximum limit + limit = min(limit, 5000) + + # Extract category name from URL + category_name = extract_category_name_from_url(category_url) + if not category_name: + return None + + # Parse wiki base URL from category URL + parsed = urlparse(category_url) + wiki_base = f"{parsed.scheme}://{parsed.netloc}" + api_url = f"{wiki_base}/w/api.php" + + # Determine MediaWiki URI + if mw_uri is None: + mw_uri = api_url + + headers = get_mediawiki_headers() + + # Build API params for category members + # cmtitle requires the full title with namespace prefix (e.g., "Category:Living_people") + params = { + "action": "query", + "list": "categorymembers", + "cmtitle": f"Category:{category_name}", + "cmtype": "page", # Only get actual pages, not subcategories or files + "cmlimit": "max", # Max per request (usually 500) + "cmnamespace": "0", # Only mainspace articles + "format": "json", + } + + articles = [] + # Seed the continue token from caller so we resume mid-category + continue_token: Optional[str] = continue_from + # Track the token that will be returned for the *next* batch + next_continue: Optional[str] = None + + while len(articles) < limit: + # Add continue token if we have one (either seeded or from prev page) + if continue_token: + params["cmcontinue"] = continue_token + elif "cmcontinue" in params: + # Remove stale key from a previous iteration that has now been cleared + del params["cmcontinue"] + + response = requests.get( + mw_uri, + params=params, + headers=headers, + timeout=MEDIAWIKI_API_TIMEOUT + ) + + if response.status_code != 200: + break + + data = response.json() + + if "error" in data: + break + + # Extract category members + members = data.get("query", {}).get("categorymembers", []) + + for member in members: + if len(articles) >= limit: + break + + title = member.get("title") + page_id = member.get("pageid") + + if title: + # Build article URL — use /wiki/ format for cleaner URLs + encoded_title = title.replace(" ", "_") + article_url = f"{wiki_base}/wiki/{encoded_title}" + + articles.append({ + "title": title, + "url": article_url, + "page_id": page_id, + }) + + # Check for continuation token from MediaWiki + continue_data = data.get("continue") + if continue_data and "cmcontinue" in continue_data: + next_continue = continue_data["cmcontinue"] + continue_token = next_continue + else: + # No more pages in the category + next_continue = None + break + + # If we hit the limit mid-category the outer while loop exits here. + # next_continue already holds the right token for resuming. + + # has_more is True when we stopped because we hit `limit` AND there are + # still more articles beyond this batch (next_continue is not None). + has_more = next_continue is not None and len(articles) >= limit + + return { + "articles": articles, + "total": len(articles), + "category": category_name, + "wiki_base": wiki_base, + # Pagination fields for "Import Next Batch" support + "has_more": has_more, + "next_continue": next_continue if has_more else None, + } + + except Exception as e: # pylint: disable=broad-exception-caught + # If crawling fails, return None to indicate failure + # Log the error for debugging + import logging # pylint: disable=import-outside-toplevel + logging.error(f"crawl_category_articles failed: {str(e)}") + return None + + diff --git a/backend/requirements.txt b/backend/requirements.txt index ac194548..ff60b2e2 100644 --- a/backend/requirements.txt +++ b/backend/requirements.txt @@ -26,6 +26,7 @@ marshmallow==3.20.2 marshmallow-sqlalchemy==0.29.0 mccabe==0.7.0 mwoauth==0.4.0 +mwparserfromhell==0.6.6 oauthlib==3.3.1 packaging==25.0 platformdirs==4.5.1 diff --git a/frontend/package-lock.json b/frontend/package-lock.json index f603902f..e38d06e7 100644 --- a/frontend/package-lock.json +++ b/frontend/package-lock.json @@ -1084,7 +1084,6 @@ "integrity": "sha512-NZyJarBfL7nWwIq+FDL6Zp/yHEhePMNnnJ0y3qfieCrmNvYct8uvtiV41UvlSe6apAfk0fY1FbWx+NwfmpvtTg==", "dev": true, "license": "MIT", - "peer": true, "bin": { "acorn": "bin/acorn" }, @@ -1881,7 +1880,6 @@ "deprecated": "This version is no longer supported. Please see https://eslint.org/version-support for other options.", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@eslint-community/eslint-utils": "^4.2.0", "@eslint-community/regexpp": "^4.6.1", @@ -2078,7 +2076,6 @@ "integrity": "sha512-whOE1HFo/qJDyX4SnXzP4N6zOWn79WhnCUY/iDR0mPfQZO8wcYE4JClzI2oZrhBnnMUCBCHZhO6VQyoBU95mZA==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@rtsao/scc": "^1.1.0", "array-includes": "^3.1.9", @@ -2136,7 +2133,6 @@ "integrity": "sha512-jDex9s7D/Qial8AGVIHq4W7NswpUD5DPDL2RH8Lzd9EloWUuvUkHfv4FRLMipH5q2UtyurorBkPeNi1wVWNh3Q==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "builtins": "^5.0.1", "eslint-plugin-es": "^4.1.0", @@ -2176,7 +2172,6 @@ "integrity": "sha512-57Zzfw8G6+Gq7axm2Pdo3gW/Rx3h9Yywgn61uE/3elTCOePEHVrn2i5CdfBwA1BLK0Q0WqctICIUSqXZW/VprQ==", "dev": true, "license": "ISC", - "peer": true, "engines": { "node": "^12.22.0 || ^14.17.0 || >=16.0.0" }, @@ -2193,7 +2188,6 @@ "integrity": "sha512-174lJKuNsuDIlLpjeXc5E2Tss8P44uIimAfGD0b90k0NoirJqpG7stLuU9Vp/9ioTOrQdWVREc4mRd1BD+CvGw==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@eslint-community/eslint-utils": "^4.4.0", "globals": "^13.24.0", @@ -4521,7 +4515,6 @@ "integrity": "sha512-o5a9xKjbtuhY6Bi5S3+HvbRERmouabWbyUcpXXUA1u+GNUKoROi9byOJ8M0nHbHYHkYICiMlqxkg1KkYmm25Sw==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "esbuild": "^0.21.3", "postcss": "^8.4.43", @@ -4581,7 +4574,6 @@ "resolved": "https://registry.npmjs.org/vue/-/vue-3.5.25.tgz", "integrity": "sha512-YLVdgv2K13WJ6n+kD5owehKtEXwdwXuj2TTyJMsO7pSeKw2bfRNZGjhB7YzrpbMYj5b5QsUebHpOqR3R3ziy/g==", "license": "MIT", - "peer": true, "dependencies": { "@vue/compiler-dom": "3.5.25", "@vue/compiler-sfc": "3.5.25", diff --git a/frontend/src/App.vue b/frontend/src/App.vue index aedc6933..2073df6a 100644 --- a/frontend/src/App.vue +++ b/frontend/src/App.vue @@ -73,7 +73,7 @@ id="userDropdown"
  • - Trusted Members + Trusted Members
  • diff --git a/frontend/src/components/AlertContainer.vue b/frontend/src/components/AlertContainer.vue index d1b7ba12..94d41fba 100644 --- a/frontend/src/components/AlertContainer.vue +++ b/frontend/src/components/AlertContainer.vue @@ -2,7 +2,7 @@
    diff --git a/frontend/src/components/ContestLeaderboard.vue b/frontend/src/components/ContestLeaderboard.vue index be29fdbd..c3b25383 100644 --- a/frontend/src/components/ContestLeaderboard.vue +++ b/frontend/src/components/ContestLeaderboard.vue @@ -7,11 +7,7 @@ Back to Contest - -
    - - -
    - -
    -
    -
    -
    - -
    -
    -
    -
    {{ contestStats.total_submissions }}
    -
    Total Submissions
    -
    -
    -
    - - -
    -
    -
    -
    - -
    -
    -
    -
    {{ contestStats.total_reviewed }}
    -
    Reviewed
    -
    -
    -
    - - -
    -
    -
    -
    - -
    -
    -
    -
    {{ contestStats.total_pending }}
    -
    Pending Review
    -
    -
    -
    - - -
    -
    -
    -
    - -
    -
    -
    -
    {{ contestStats.total_marks_awarded }}
    -
    Total Marks
    -
    -
    -
    -
    - - - -
    -
    -
    Filters
    -
    -
    -
    - -
    - - -
    - - -
    - - -
    - - -
    - - -
    - - -
    - -
    -
    -
    -
    - - -
    +

    No participants found

    -

    - {{ filters.filter_type !== 'all' || filters.min_marks - ? 'Try adjusting your filters to see more results' - : 'No submissions have been made to this contest yet' }} -

    +

    No submissions have been made to this contest yet

    @@ -179,9 +59,8 @@ class="form-control"
    - Rankings + Participants
    - {{ pagination.total_results }} participants
    @@ -190,7 +69,6 @@ class="form-control" - @@ -199,19 +77,7 @@ class="form-control" - - - - +
    Rank Username Submissions Total Marks
    -
    - - - - {{ participant.rank }} -
    -
    @@ -240,33 +106,21 @@ class="form-control"
    -