File "parse_pakmcqs.py"
Full path: /home/algopkco/public_html/scraper/parse_pakmcqs.py
File
size: 6.34 B (6.34 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8
Download Open Edit Advanced Editor &nnbsp; Back
import hashlib
import html
import json
import os
import re
import sys
import time
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
BASE = os.path.dirname(os.path.abspath(__file__))
POSTS_FILE = os.path.join(BASE, "data", "pakmcqs_posts.jsonl")
CATS_FILE = os.path.join(BASE, "data", "pakmcqs_categories.json")
OUT_FILE = os.path.join(BASE, "data", "parsed_mcqs.jsonl")
IMG_TAG_RE = re.compile(r"<img[^>]*>")
BR_RE = re.compile(r"<br\s*/?>", re.I)
P_OPEN_RE = re.compile(r"<p[^>]*>")
P_CLOSE_RE = re.compile(r"</p>")
TAG_RE = re.compile(r"<[^>]+>")
WS_RE = re.compile(r"\s+")
GARBAGE_RE = re.compile(
r"(Submitted by.*|Advertisement|Correct Answer|The correct answer to the question|"
r"Read More Details about this Mcq|Leave a Comment|Comments are closed|"
r"the correct answer is|mcqshubs|PakMcqs)", re.I
)
def clean_text(s):
s = html.unescape(s or "")
s = WS_RE.sub(" ", s)
return s.strip()
def parse_options_and_answer(content_html):
"""Extract options list and index of correct option (1-based) from post content."""
h = content_html or ""
more_idx = h.find("<!--more")
if more_idx > 0:
h = h[:more_idx]
sub_idx = h.lower().find("submitted by")
if sub_idx > 0:
h = h[:sub_idx]
img_removed = IMG_TAG_RE.sub("", h)
strong_flags = []
for m in re.finditer(r"<strong[^>]*>(.*?)</strong>", img_removed, re.S):
strong_flags.append(clean_text(m.group(1)))
block = TAG_RE.sub(" ", img_removed)
block = html.unescape(block)
block = WS_RE.sub(" ", block)
m = re.search(
r"(?:^|\s)([A-H])\.\s*(.*?)(?=\s+([A-H])\.\s*|$)",
block,
re.S,
)
letters = re.findall(r"(?:^|\s)([A-H])\.\s", block)
if not letters:
return None, None, None
seen = []
for l in letters:
if l not in seen:
seen.append(l)
if len(seen) < 2:
return None, None, None
opts = []
for idx, letter in enumerate(seen):
start = block.find(letter + ".")
if idx + 1 < len(seen):
end = block.find(seen[idx + 1] + ".")
else:
end = len(block)
if start < 0 or end < start:
return None, None, None
opt_text = clean_text(block[start + 2 : end])
if not opt_text:
return None, None, None
opts.append(opt_text)
correct_idx = None
for i, opt in enumerate(opts):
for flag in strong_flags:
stripped = re.sub(r"^[A-H]\.\s+", "", flag)
cands = {flag, stripped}
for cand in cands:
if cand and (opt == cand or opt.startswith(cand) or cand in opt):
correct_idx = i + 1
break
if correct_idx:
break
if correct_idx:
break
return opts, correct_idx, None
def extract_explanation(content_html, options=None):
"""Explanation = text after 'Submitted by'/'Updated by' or after options block."""
h = content_html or ""
m = re.search(r"(Submitted|Updated)\s+by\s*:\s*(<[^>]*>)*\s*[^<\n]*?(</strong>|</p>|$)", h, re.I | re.S)
if m:
h = h[m.end():]
else:
letters = re.findall(r"(?:^|\s)([A-H])\.\s", re.sub(r"<[^>]+>", " ", h))
if not letters:
h = ""
h = IMG_TAG_RE.sub("", h)
h = re.sub(r"<!--.*?-->", " ", h, flags=re.S)
txt = TAG_RE.sub(" ", h)
txt = html.unescape(txt)
txt = WS_RE.sub(" ", txt)
if options:
for opt in options:
if opt:
txt = txt.replace(opt, "", 1)
txt = re.sub(r"^\s*(?:[A-H]\.\s*)+", "", txt)
txt = GARBAGE_RE.sub(" ", txt)
txt = WS_RE.sub(" ", txt)
return clean_text(txt)
def norm(text):
t = html.unescape(text or "")
t = re.sub(r"[^a-z0-9]", "", t.lower())
return t
def parse_all():
cats = {}
if os.path.exists(CATS_FILE):
with open(CATS_FILE, "r", encoding="utf-8") as f:
for ln in f:
try:
c = json.loads(ln)
cats[c["id"]] = c
except Exception:
pass
parent_chain = {}
for cid, c in cats.items():
chain = []
cur = c
guard = 0
while cur and cur["parent"] and guard < 10:
cur = cats.get(cur["parent"])
guard += 1
if cur:
chain.append(cur["id"])
parent_chain[cid] = chain
n_ok = 0
n_skip = 0
seen_q = {}
parsed = []
with open(POSTS_FILE, "r", encoding="utf-8") as f:
for ln in f:
try:
p = json.loads(ln)
except Exception:
continue
title = clean_text(p["title"].get("rendered", ""))
content_html = p["content"].get("rendered", "")
if not title:
n_skip += 1
continue
opts, correct_idx, _ = parse_options_and_answer(content_html)
if not opts or not correct_idx or len(opts) < 2 or correct_idx > len(opts):
n_skip += 1
continue
q = clean_text(title)
key = norm(q)
if key in seen_q:
n_skip += 1
continue
seen_q[key] = True
cat_ids = p.get("categories", []) or []
primary = cat_ids[0] if cat_ids else None
chain = parent_chain.get(primary, []) if primary else []
cat_name = cats.get(primary, {}).get("name", "General") if primary else "General"
parent_name = cats.get(chain[0], {}).get("name", "") if chain else ""
parsed.append(
{
"post_id": p["id"],
"question": q,
"options": opts,
"correct_index": correct_idx,
"explanation": extract_explanation(content_html, opts),
"category_id": primary,
"category": cat_name,
"parent_category": parent_name,
"chain": chain,
}
)
n_ok += 1
with open(OUT_FILE, "w", encoding="utf-8") as f:
for item in parsed:
f.write(json.dumps(item, ensure_ascii=False) + "\n")
print("parsed OK:", n_ok, "skipped:", n_skip)
return parsed
if __name__ == "__main__":
t0 = time.time()
parse_all()
print("time:", round(time.time() - t0, 1), "s")