File "parse_pakmcqs.py"

Full path: /home/algopkco/public_html/scraper/parse_pakmcqs.py
File size: 6.34 B (6.34 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8

Download   Open   Edit   Advanced Editor &nnbsp; Back

import hashlib
import html
import json
import os
import re
import sys
import time

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))

BASE = os.path.dirname(os.path.abspath(__file__))
POSTS_FILE = os.path.join(BASE, "data", "pakmcqs_posts.jsonl")
CATS_FILE = os.path.join(BASE, "data", "pakmcqs_categories.json")
OUT_FILE = os.path.join(BASE, "data", "parsed_mcqs.jsonl")

IMG_TAG_RE = re.compile(r"<img[^>]*>")
BR_RE = re.compile(r"<br\s*/?>", re.I)
P_OPEN_RE = re.compile(r"<p[^>]*>")
P_CLOSE_RE = re.compile(r"</p>")
TAG_RE = re.compile(r"<[^>]+>")
WS_RE = re.compile(r"\s+")
GARBAGE_RE = re.compile(
    r"(Submitted by.*|Advertisement|Correct Answer|The correct answer to the question|"
    r"Read More Details about this Mcq|Leave a Comment|Comments are closed|"
    r"the correct answer is|mcqshubs|PakMcqs)", re.I
)


def clean_text(s):
    s = html.unescape(s or "")
    s = WS_RE.sub(" ", s)
    return s.strip()


def parse_options_and_answer(content_html):
    """Extract options list and index of correct option (1-based) from post content."""
    h = content_html or ""
    more_idx = h.find("<!--more")
    if more_idx > 0:
        h = h[:more_idx]
    sub_idx = h.lower().find("submitted by")
    if sub_idx > 0:
        h = h[:sub_idx]

    img_removed = IMG_TAG_RE.sub("", h)
    strong_flags = []
    for m in re.finditer(r"<strong[^>]*>(.*?)</strong>", img_removed, re.S):
        strong_flags.append(clean_text(m.group(1)))

    block = TAG_RE.sub(" ", img_removed)
    block = html.unescape(block)
    block = WS_RE.sub(" ", block)

    m = re.search(
        r"(?:^|\s)([A-H])\.\s*(.*?)(?=\s+([A-H])\.\s*|$)",
        block,
        re.S,
    )
    letters = re.findall(r"(?:^|\s)([A-H])\.\s", block)
    if not letters:
        return None, None, None
    seen = []
    for l in letters:
        if l not in seen:
            seen.append(l)
    if len(seen) < 2:
        return None, None, None

    opts = []
    for idx, letter in enumerate(seen):
        start = block.find(letter + ".")
        if idx + 1 < len(seen):
            end = block.find(seen[idx + 1] + ".")
        else:
            end = len(block)
        if start < 0 or end < start:
            return None, None, None
        opt_text = clean_text(block[start + 2 : end])
        if not opt_text:
            return None, None, None
        opts.append(opt_text)

    correct_idx = None
    for i, opt in enumerate(opts):
        for flag in strong_flags:
            stripped = re.sub(r"^[A-H]\.\s+", "", flag)
            cands = {flag, stripped}
            for cand in cands:
                if cand and (opt == cand or opt.startswith(cand) or cand in opt):
                    correct_idx = i + 1
                    break
            if correct_idx:
                break
        if correct_idx:
            break
    return opts, correct_idx, None


def extract_explanation(content_html, options=None):
    """Explanation = text after 'Submitted by'/'Updated by' or after options block."""
    h = content_html or ""
    m = re.search(r"(Submitted|Updated)\s+by\s*:\s*(<[^>]*>)*\s*[^<\n]*?(</strong>|</p>|$)", h, re.I | re.S)
    if m:
        h = h[m.end():]
    else:
        letters = re.findall(r"(?:^|\s)([A-H])\.\s", re.sub(r"<[^>]+>", " ", h))
        if not letters:
            h = ""
    h = IMG_TAG_RE.sub("", h)
    h = re.sub(r"<!--.*?-->", " ", h, flags=re.S)
    txt = TAG_RE.sub(" ", h)
    txt = html.unescape(txt)
    txt = WS_RE.sub(" ", txt)
    if options:
        for opt in options:
            if opt:
                txt = txt.replace(opt, "", 1)
    txt = re.sub(r"^\s*(?:[A-H]\.\s*)+", "", txt)
    txt = GARBAGE_RE.sub(" ", txt)
    txt = WS_RE.sub(" ", txt)
    return clean_text(txt)


def norm(text):
    t = html.unescape(text or "")
    t = re.sub(r"[^a-z0-9]", "", t.lower())
    return t


def parse_all():
    cats = {}
    if os.path.exists(CATS_FILE):
        with open(CATS_FILE, "r", encoding="utf-8") as f:
            for ln in f:
                try:
                    c = json.loads(ln)
                    cats[c["id"]] = c
                except Exception:
                    pass
    parent_chain = {}
    for cid, c in cats.items():
        chain = []
        cur = c
        guard = 0
        while cur and cur["parent"] and guard < 10:
            cur = cats.get(cur["parent"])
            guard += 1
            if cur:
                chain.append(cur["id"])
        parent_chain[cid] = chain

    n_ok = 0
    n_skip = 0
    seen_q = {}
    parsed = []
    with open(POSTS_FILE, "r", encoding="utf-8") as f:
        for ln in f:
            try:
                p = json.loads(ln)
            except Exception:
                continue
            title = clean_text(p["title"].get("rendered", ""))
            content_html = p["content"].get("rendered", "")
            if not title:
                n_skip += 1
                continue
            opts, correct_idx, _ = parse_options_and_answer(content_html)
            if not opts or not correct_idx or len(opts) < 2 or correct_idx > len(opts):
                n_skip += 1
                continue
            q = clean_text(title)
            key = norm(q)
            if key in seen_q:
                n_skip += 1
                continue
            seen_q[key] = True
            cat_ids = p.get("categories", []) or []
            primary = cat_ids[0] if cat_ids else None
            chain = parent_chain.get(primary, []) if primary else []
            cat_name = cats.get(primary, {}).get("name", "General") if primary else "General"
            parent_name = cats.get(chain[0], {}).get("name", "") if chain else ""
            parsed.append(
                {
                    "post_id": p["id"],
                    "question": q,
                    "options": opts,
                    "correct_index": correct_idx,
                    "explanation": extract_explanation(content_html, opts),
                    "category_id": primary,
                    "category": cat_name,
                    "parent_category": parent_name,
                    "chain": chain,
                }
            )
            n_ok += 1

    with open(OUT_FILE, "w", encoding="utf-8") as f:
        for item in parsed:
            f.write(json.dumps(item, ensure_ascii=False) + "\n")
    print("parsed OK:", n_ok, "skipped:", n_skip)
    return parsed


if __name__ == "__main__":
    t0 = time.time()
    parse_all()
    print("time:", round(time.time() - t0, 1), "s")