File "fetch_t4cpp.py"

Full path: /home/algopkco/public_html/scraper/fetch_t4cpp.py
File size: 5.48 B (5.48 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8

Download   Open   Edit   Advanced Editor &nnbsp; Back

import html as H
import json
import os
import re
import time

import requests

BASE = os.path.dirname(os.path.abspath(__file__))
DATA = os.path.join(BASE, "data")
OUT = os.path.join(DATA, "t4cpp_mcqs.jsonl")
PROG = os.path.join(BASE, "t4cpp_progress.txt")
LOG = os.path.join(BASE, "t4cpp_run.log")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36",
}
s = requests.Session()
s.headers.update(HEADERS)

CPP_PAGES = [
    "advanced-c-plus-plus-mcqs",
    "arrays-mcqs-questions-answers-c",
    "c-array-solved-mcqs-questions-answers",
    "c-standard-library-mcqs-questions-answers",
    "classes-and-inheritance-mcqs-in-c-oop",
    "friend-function-mcqs",
    "highly-recommended-c-important-mcqs-with-explanation",
    "inline-function-mcqs",
    "mcqs-of-introduction-to-programming",
    "parameter-passing-mechanisms-call-by-value-referencemcqs",
    "pointers-solved-mcqs-questions-answers",
    "polymorphism-mcqs-in-object-oriented-programmingoop",
    "polymorphism-mcqs-viva-questions",
    "programming-c-mcqs-for-programming-competition",
    "top-50-programming-c-solved-mcqs-questions-answers",
    "virtual-function-mcqs-c",
]

SIDEBAR_EXTRA = [
    "low-level-and-high-level-languages-mcqs",
    "procedural-and-non-procedural-languages-mcqs",
    "arrays-mcqs-2",
    "one-dimensional-arrays-mcqs",
    "multi-dimensional-arrays-mcqs",
    "oop-intro-and-examples-mcqs",
    "mcqs-of-oop",
    "operator-overloading-mcqs",
    "cpp-past-papers-2022-mcqs",
    "cpp-past-papers-2021-mcqs",
    "cpp-past-papers-2020-mcqs",
    "cpp-past-papers-2019-mcqs",
    "best-c-interview-questions-and-answers",
    "c-important-mcqs",
]

KEYWORDS = re.compile(r"(mcq|c\+\+|cplus|cpp|oop|passing|cpp-past|pointer|array|inline|virtual|friend|polymorph)", re.I)


def norm(t):
    t = H.unescape(t)
    t = re.sub(r"<[^>]+>", "", t)
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def get(url, tries=4):
    for i in range(tries):
        try:
            r = s.get(url, timeout=30)
            if r.status_code == 200:
                return r.text
        except Exception as e:
            print("err", e, "retry", i, flush=True)
        time.sleep(1 + i)
    return None


def parse_pages(html, url):
    rows = []
    blocks = re.findall(r'<div class="question-container">(.*?)</div>', html, re.S)
    for b in blocks:
        qm = re.search(r'question-text">(.*?)</strong>', b, re.S)
        if not qm:
            continue
        q = norm(qm.group(1))
        q = re.sub(r"^\d+\.\s*", "", q)
        if not q or len(q) < 10:
            continue
        labels = re.findall(r'<label>(.*?)</label>', b, re.S)
        opts = []
        correct_text = None
        for lb in labels:
            vm = re.search(r'value="(.*?)"', lb, re.S)
            om = re.search(r"onclick=\"checkAnswer\('[^']*', ?'(.*?)'\)\"", lb, re.S)
            if not vm:
                continue
            if om and correct_text is None:
                correct_text = norm(om.group(1))
            label = norm(lb)
            m = re.search(r"\(([A-E])\)\s*(.*)$", label, re.S)
            if m:
                opts.append(norm(m.group(2)))
            else:
                opts.append(norm(vm.group(1)))
        if len(opts) < 2:
            continue
        ci = -1
        if correct_text:
            for i, o in enumerate(opts):
                if norm(o) == norm(correct_text):
                    ci = i
                    break
        if ci < 0:
            continue
        rows.append({
            "subject": "C++ Programming",
            "topic": "T4Tutorials C++ MCQs",
            "question": q,
            "options": opts,
            "correct": ci + 1,
            "explanation": "",
            "source_url": url,
        })
    return rows


def main():
    done = set()
    if os.path.exists(PROG):
        done = set(l.strip() for l in open(PROG, encoding="utf-8") if l.strip())
    seen = set()
    if os.path.exists(OUT):
        for ln in open(OUT, encoding="utf-8"):
            try:
                seen.add(json.loads(ln)["question"])
            except Exception:
                pass
    fout = open(OUT, "a", encoding="utf-8")
    flog = open(LOG, "a", encoding="utf-8")
    new = 0
    queue = list(CPP_PAGES + SIDEBAR_EXTRA)
    visited = set()
    while queue:
        slug = queue.pop(0)
        if slug in visited or slug in done:
            continue
        visited.add(slug)
        url = f"https://t4tutorials.com/{slug}/"
        html = get(url)
        if not html:
            flog.write(f"{slug}: NO HTML\n")
            flog.flush()
            continue
        rows = parse_pages(html, url)
        extra = set(re.findall(r'href="https://t4tutorials\.com/([a-z0-9\-]+)/?"', html))
        for e in extra:
            if KEYWORDS.search(e) and "mcq" in e and e not in visited and e not in done:
                queue.append(e)
        cnt = 0
        for it in rows:
            if it["question"] in seen:
                continue
            seen.add(it["question"])
            fout.write(json.dumps(it, ensure_ascii=False) + "\n")
            cnt += 1
            new += 1
        fout.flush()
        print(f"{slug}: {len(rows)} blocks, {cnt} new (total new {new})", flush=True)
        flog.write(f"{slug}: {len(rows)} parsed, {cnt} new\n")
        flog.flush()
        with open(PROG, "a", encoding="utf-8") as f:
            f.write(slug + "\n")
        time.sleep(0.5)
    fout.close()
    flog.close()
    print("DONE total new:", new)


if __name__ == "__main__":
    main()