File "fetch_mcqsets_cpp.py"

Full path: /home/algopkco/public_html/scraper/fetch_mcqsets_cpp.py
File size: 3.97 B (3.97 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8

Download   Open   Edit   Advanced Editor &nnbsp; Back

import json
import os
import re
import time

import requests

BASE = os.path.dirname(os.path.abspath(__file__))
DATA = os.path.join(BASE, "data")
OUT = os.path.join(DATA, "mcqsets_cpp_mcqs.jsonl")
PROG = os.path.join(BASE, "mcqsets_progress.txt")
LOG = os.path.join(BASE, "mcqsets_run.log")

H = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/126.0 Safari/537.36"}
s = requests.Session()
s.headers.update(H)

SEED = "https://mcqsets.com/mcq-questions-2/c/solved-c-plus-mcq-set/"


def get(url, tries=3):
    for i in range(tries):
        try:
            r = s.get(url, timeout=30)
            if r.status_code == 200:
                return r.text
        except Exception:
            pass
        time.sleep(1 + i)
    return None


def parse(html):
    body = re.sub(r"<script.*?</script>|<style.*?</style>", "", html, flags=re.S)
    plain = re.sub(r"<[^>]+>", " ", body)
    plain = re.sub(r"\s+", " ", plain)
    plain = re.sub(r"&#?\w+;", " ", plain)
    qs = []
    for m in re.finditer(
        r"(\d+)\.\s+(.{15,}?)\s+a\.\s+(.{1,300}?)\s+b\.\s+(.{1,300}?)\s+c\.\s+(.{1,300}?)\s+d\.\s+(.{1,300}?)(?=\s+\d+\.|Answers? to|Filed Under|\u00a0|&nbsp)",
        plain,
    ):
        qs.append({
            "num": int(m.group(1)),
            "q": re.sub(r"^\s*\d+\s*", "", m.group(2)).strip(),
            "opts": [m.group(3).strip(), m.group(4).strip(), m.group(5).strip(), m.group(6).strip()],
        })
    ans = {}
    am = re.search(r"Answers? to[^0-9]{0,120}?((?:\d+\s*[\-\u2011\u2013\u2014]?\s*[a-eA-E]\s*)+)", plain)
    if am:
        for pair in re.findall(r"(\d+)\s*[\-\u2011\u2013\u2014]?\s*([a-eA-E])", am.group(1)):
            ans[int(pair[0])] = pair[1].lower()
    rows = []
    for item in qs:
        let = ans.get(item["num"])
        if not let:
            continue
        ci = "abcde".find(let)
        if ci < 0 or ci >= len(item["opts"]):
            continue
        if len(item["q"]) < 10 or any(len(o) < 1 for o in item["opts"]):
            continue
        rows.append((item["num"], item["q"], item["opts"], ci))
    return rows


def main():
    done = set()
    if os.path.exists(PROG):
        done = set(l.strip() for l in open(PROG, encoding="utf-8") if l.strip())
    seen = set()
    if os.path.exists(OUT):
        for ln in open(OUT, encoding="utf-8"):
            try:
                seen.add(json.loads(ln)["question"])
            except Exception:
                pass
    fout = open(OUT, "a", encoding="utf-8")
    flog = open(LOG, "a", encoding="utf-8")
    new = 0
    queue = [SEED]
    visited = set()
    while queue:
        url = queue.pop(0)
        if url in visited or url in done:
            continue
        visited.add(url)
        html = get(url)
        if not html:
            flog.write(f"{url}: NO HTML\n")
            flog.flush()
            continue
        rows = parse(html)
        extra = re.findall(r'href="(/mcq-questions-2/c/[^"#]+)"', html)
        for e in extra:
            u = "https://mcqsets.com" + e
            if "feed" in u or u in visited or u in done:
                continue
            queue.append(u)
        cnt = 0
        for num, q, opts, ci in rows:
            if q in seen:
                continue
            seen.add(q)
            fout.write(json.dumps({
                "subject": "C++ Programming",
                "topic": "mcqsets C++ Sets",
                "question": q,
                "options": opts,
                "correct": ci + 1,
                "explanation": "",
                "source_url": url,
            }, ensure_ascii=False) + "\n")
            cnt += 1
            new += 1
        fout.flush()
        print(f"{url.split('/')[-2]}: {len(rows)} parsed, {cnt} new (total new {new})", flush=True)
        flog.write(f"{url.split('/')[-2]}: {len(rows)} parsed, {cnt} new\n")
        flog.flush()
        with open(PROG, "a", encoding="utf-8") as f:
            f.write(url + "\n")
        time.sleep(0.6)
    fout.close()
    flog.close()
    print("DONE total new:", new)


if __name__ == "__main__":
    main()