File "fetch_mcqsets_more.py"

Full path: /home/algopkco/public_html/scraper/fetch_mcqsets_more.py
File size: 5.18 B (5.18 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8

Download   Open   Edit   Advanced Editor &nnbsp; Back

import json
import os
import re
import time

import requests

BASE = os.path.dirname(os.path.abspath(__file__))
DATA = os.path.join(BASE, "data")
OUT = os.path.join(DATA, "mcqsets_more_mcqs.jsonl")
PROG = os.path.join(BASE, "mcqsets_more_progress.txt")
LOG = os.path.join(BASE, "mcqsets_more_run.log")

H = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/126.0 Safari/537.36"}
s = requests.Session()
s.headers.update(H)

# prefix, subject, topic
CATS = [
    ("mcq-questions-2/c", "C Programming", "mcqsets C Sets"),
    ("mcq-questions-2/data-structures-and-algorithms", "Data Structures And Algorithms", "mcqsets DSA"),
    ("mcq-questions-2/computer-fundamentals", "Computer Science General Test", "mcqsets Fundamentals"),
    ("mcq-questions-2/operating-system", "Operating Systems", "mcqsets OS"),
    ("mcq-questions-2/html-web-page-designing", "Web Technologies", "mcqsets HTML"),
    ("mcq-questions-2/microsoft-word", "Computer Science General Test", "mcqsets MS Word"),
    ("mcq-questions-2/ms-excel", "Computer Science General Test", "mcqsets MS Excel"),
    ("mcq-questions-2/ms-powerpoint", "Computer Science General Test", "mcqsets MS PowerPoint"),
    ("mcq-questions-2/ms-access-mcq-questions-2", "Database Management Systems", "mcqsets MS Access"),
    ("mcq-questions-collection", "Computer Science General Test", "mcqsets Collection"),
]

FORMAT_LOWER = re.compile(
    r"(\d+)\.\s+(.{15,}?)\s+a\.\s+(.{1,300}?)\s+b\.\s+(.{1,300}?)\s+c\.\s+(.{1,300}?)\s+d\.\s+(.{1,300}?)(?=\s+\d+\.|Answers? to|Filed Under|&nbsp|$)",
    re.S,
)
FORMAT_UPPER = re.compile(
    r"(\d+)\.\s+(.{15,}?)\s+A\)\s+(.{1,300}?)\s+B\)\s+(.{1,300}?)\s+C\)\s+(.{1,300}?)\s+D\)\s+(.{1,300}?)(?=\s+\d+\.|Answers? to|Filed Under|&nbsp|$)",
    re.S,
)
ANS_KEY = re.compile(r"Answers? to[^0-9]{0,120}?((?:\d+\s*[\-\u2011\u2013\u2014]?\s*[a-eA-E]\s*)+)")


def get(url, tries=3):
    for i in range(tries):
        try:
            r = s.get(url, timeout=30)
            if r.status_code == 200:
                return r.text
        except Exception:
            pass
        time.sleep(1 + i)
    return None


def parse(html):
    body = re.sub(r"<script.*?</script>|<style.*?</style>", "", html, flags=re.S)
    plain = re.sub(r"<[^>]+>", " ", body)
    plain = re.sub(r"\s+", " ", plain)
    plain = re.sub(r"&#?\w+;", " ", plain)
    qs = []
    for m in FORMAT_LOWER.finditer(plain):
        qs.append((int(m.group(1)), m.group(2).strip(), [m.group(3).strip(), m.group(4).strip(), m.group(5).strip(), m.group(6).strip()]))
    if not qs:
        for m in FORMAT_UPPER.finditer(plain):
            qs.append((int(m.group(1)), m.group(2).strip(), [m.group(3).strip(), m.group(4).strip(), m.group(5).strip(), m.group(6).strip()]))
    ans = {}
    am = ANS_KEY.search(plain)
    if am:
        for pair in re.findall(r"(\d+)\s*[\-\u2011\u2013\u2014]?\s*([a-eA-E])", am.group(1)):
            ans[int(pair[0])] = pair[1].lower()
    rows = []
    for num, q, opts in qs:
        let = ans.get(num)
        if not let:
            continue
        ci = "abcde".find(let)
        if ci < 0 or ci >= len(opts):
            continue
        if len(q) < 10:
            continue
        rows.append((q, opts, ci))
    return rows


def main():
    done = set()
    if os.path.exists(PROG):
        done = set(l.strip() for l in open(PROG, encoding="utf-8") if l.strip())
    seen = set()
    if os.path.exists(OUT):
        for ln in open(OUT, encoding="utf-8"):
            try:
                seen.add(json.loads(ln)["question"])
            except Exception:
                pass
    fout = open(OUT, "a", encoding="utf-8")
    flog = open(LOG, "a", encoding="utf-8")
    new = 0
    for path, subj, topic in CATS:
        url = f"https://mcqsets.com/{path}/"
        html = get(url)
        if not html:
            flog.write(f"{path}: NO HTML\n")
            flog.flush()
            continue
        slugs = sorted(set(re.findall(r'href="/(?:' + re.escape(path) + r')/([a-z0-9\-]+)/?"', html)))
        queue = [f"/{path}/" + sl + "/" for sl in slugs]
        queue.insert(0, url)
        for u in queue:
            if u in done:
                continue
            h = get(u)
            if not h:
                flog.write(f"{u}: NO HTML\n")
                flog.flush()
                with open(PROG, "a", encoding="utf-8") as f:
                    f.write(u + "\n")
                continue
            rows = parse(h)
            cnt = 0
            for q, opts, ci in rows:
                if q in seen:
                    continue
                seen.add(q)
                fout.write(json.dumps({
                    "subject": subj, "topic": topic, "question": q,
                    "options": opts, "correct": ci + 1, "explanation": "", "source_url": u,
                }, ensure_ascii=False) + "\n")
                cnt += 1
                new += 1
            fout.flush()
            print(f"{u}: {len(rows)} parsed, {cnt} new (total {new})", flush=True)
            flog.write(f"{u}: {len(rows)} parsed, {cnt} new\n")
            flog.flush()
            with open(PROG, "a", encoding="utf-8") as f:
                f.write(u + "\n")
            time.sleep(0.5)
    fout.close()
    flog.close()
    print("DONE total new:", new)


if __name__ == "__main__":
    main()