File "fetch_t4cpp2.py"

Full path: /home/algopkco/public_html/scraper/fetch_t4cpp2.py
File size: 10.64 B
MIME-type: text/x-script.python
Charset: utf-8

Download   Open   Edit   Advanced Editor &nnbsp; Back

import html as H
import json
import os
import re
import time

import requests

BASE = os.path.dirname(os.path.abspath(__file__))
DATA = os.path.join(BASE, "data")
OUT = os.path.join(DATA, "t4_mcqs.jsonl")
PROG = os.path.join(BASE, "t4_progress2.txt")
LOG = os.path.join(BASE, "t4_run2.log")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36",
}
s = requests.Session()
s.headers.update(HEADERS)

SLUGS = [l.strip() for l in open(os.path.join(BASE, "t4cpp_progress.txt"), encoding="utf-8") if l.strip()]

CPP_SLUGS = {
    "advanced-c-plus-plus-mcqs", "arrays-mcqs-questions-answers-c", "c-array-solved-mcqs-questions-answers",
    "c-mcqs", "c-standard-library-mcqs-questions-answers", "classes-and-inheritance-mcqs-in-c-oop",
    "friend-function-mcqs", "function-arguments-pass-by-value-reference-c-mcqs", "inline-function-mcqs",
    "mcqs-of-introduction-to-programming", "multi-dimensional-arrays-mcqs",
    "one-dimensional-and-multi-dimensional-arrays-c-mcqs",
    "one-dimensional-arrays-mcqs", "operator-overloading-solved-mcqs-oop",
    "parameter-passing-mechanisms-call-by-value-referencemcqs",
    "pointers-and-arrays-c-mcqs", "pointers-solved-mcqs-questions-answers", "pointers-to-functions-c-mcqs",
    "pointers-to-pointers-c-mcqs", "polymorphism-mcqs-in-object-oriented-programmingoop",
    "programming-c-mcqs-for-programming-competition",
}

CS_SUBJECTS = {
    "automata-theory-mcqs": "Computer Science General Test",
    "compiler-construction-mcqs": "Computer Science General Test",
    "computer-graphics-solved-mcqs-questions-answers": "Computer Science General Test",
    "cyber-crime-solved-mcqs-questions-answers": "Computer Science General Test",
    "deadlock-mcqs-questions-answers": "Operating Systems",
    "digital-image-processing-mcqs": "Computer Science General Test",
    "distributed-database-architecture-mcqs-mcqs-in-dbms": "Database Management Systems",
    "html-mcqs-test-solved-questions-answers": "Web Technologies",
    "html-solved-mcqs": "Web Technologies",
    "css-mcqs-for-web-developer-and-managers-jobs-test": "Web Technologies",
    "introduction-to-computing-itc-mcqs": "Computer Science General Test",
    "java-online-test-mcqs-questions-answers": "Java Programming",
    "management-information-system-mis-solved-mcqs-with-answers-pdf": "Management Information Systems",
    "mcqs-analysis-of-algorithms-for-jobs-test-solved": "Design And Analysis Of Algorithms",
    "mcqs-computer-science-core-courses": "Computer Science General Test",
    "mcqs-elective-courses-computer-science": "Computer Science General Test",
    "mcqs-on-viruses-and-computer-security": "Computer Security",
    "mobile-android-applications-mcqs": "Mobile Application Development",
    "mpi-mcqs-message-passing-interface-mcqs": "Computer Science General Test",
    "network-layer-osi-model-solved-mcqs": "Computer Networks",
    "networking-mcqs-storage-solutions-cloud-computing-mcqs-data-center-technologies-mcqs": "Computer Networks",
    "php-mcqs-solved-questions-answers-for-web-developers-and-managers": "Web Technologies",
    "social-networks-mcqs-solved-questions-answers": "Computer Science General Test",
    "system-programming-mcqs": "Computer Science General Test",
    "virtual-memory-mcqs-questions-answers-in-operating-systems": "Operating Systems",
    "technical-report-writing-solved-mcqs": "Computer Science General Test",
    "technology-management-mcqs": "Computer Science General Test",
    "computer-science-mcqs-homepage": "Computer Science General Test",
    "computer-science-mcqs-leaks-pdf-ebook-by-fazal-rehman-shamil": "Computer Science General Test",
    "software-engineering-mcqs": "Software Engineering",
}

CS_KEYWORDS = re.compile(
    r"(computer|software|hardware|programming|algorithm|data|network|database|dbms|html|css|php|java|javascript|"
    r"python|android|operating|os-|web|sql|compiler|automata|digital|security|graphic|system|linux|cloud|"
    r"information|multimedia|micro|processor|artificial|cryptography|e-?commerce|oop|mcq-of|ic3|ict|itc|pract|"
    r"mobile|server|stack|queue|linked|sorting|searching|binary|recursion|hci|cpu|memory|input-output|"
    r"windows|ms-office|office)",
    re.I,
)


def norm(t):
    t = H.unescape(t)
    t = re.sub(r"<[^>]+>", "", t)
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def get(url, tries=4):
    for i in range(tries):
        try:
            r = s.get(url, timeout=30)
            if r.status_code == 200:
                return r.text
        except Exception:
            pass
        time.sleep(1 + i)
    return None


def parse_format_a(html):
    rows = []
    for b in re.findall(r'<div class="question-container">(.*?)</div>', html, re.S):
        qm = re.search(r"question-text\">(.*?)</strong>", b, re.S)
        if not qm:
            continue
        q = re.sub(r"^\d+\.\s*", "", norm(qm.group(1)))
        if len(q) < 10 or q.endswith("Back"):
            continue
        opts, ci = [], -1
        for lb in re.findall(r"<label>(.*?)</label>", b, re.S):
            vm = re.search(r'value="(.*?)"', lb, re.S)
            om = re.search(r"onclick=\"checkAnswer\('[^']*', ?'(.*?)'\)\"", lb, re.S)
            if not vm:
                continue
            label = norm(lb)
            m = re.search(r"\(([A-E])\)\s*(.*)$", label, re.S)
            if m:
                opts.append(norm(m.group(2)))
            else:
                opts.append(norm(vm.group(1)))
            if om and ci < 0:
                ans = norm(om.group(1))
                for i, o in enumerate(opts):
                    if o == ans:
                        ci = i
                        break
        if len(opts) >= 2 and 0 <= ci < len(opts):
            rows.append((q, opts, ci))
    return rows


def parse_format_c(html):
    rows = []
    for ch in re.split(r"Q#\s*\d+\s*:", html)[1:]:
        ans_part = None
        if "Answer:" in ch:
            body, ans_part = ch.split("Answer:", 1)
        else:
            body = ch
        pieces = re.split(r"\(([A-E])\)", body)
        q = norm(pieces[0])
        q = re.split(r"Answer:", q)[0]
        if len(q) < 5:
            continue
        opts = []
        for i in range(1, len(pieces) - 1, 2):
            opts.append(norm(pieces[i + 1]))
        opts = [o for o in opts if o]
        ci = -1
        if ans_part:
            am = re.search(r"^\s*\(([A-E])\)", ans_part)
            if am:
                ci = "ABCDE".find(am.group(1))
        if len(opts) >= 2 and 0 <= ci < len(opts):
            rows.append((q, opts, ci))
    return rows


def parse_format_d(html):
    clean = re.sub(r"<script.*?</script>|<style.*?</style>", "", html, flags=re.S)
    clean = re.sub(r"<[^>]+>", " ", clean)
    clean = H.unescape(clean)
    clean = re.sub(r"\s+", " ", clean)
    rows = []
    spans = []
    for m in re.finditer(r"Answer\s*:\s*([a-eA-E])\s*\)", clean):
        spans.append((m.start(), m.group(1)))
    for i, (pos, letter) in enumerate(spans):
        end = spans[i + 1][0] if i + 1 < len(spans) else min(pos + 400, len(clean))
        seg = clean[max(pos - 900, 0):end]
        qm = re.search(r"([A-Za-z0-9][^A-Za-z]{0,30}\?)\s", seg)
        if not qm:
            continue
        q = norm(qm.group(1))
        if len(q) < 10:
            continue
        tail = seg[qm.end():]
        opts = [norm(x) for x in re.split(r"[a-eA-E]\s*\)\s*", tail) if norm(x)]
        ci = "abcde".find(letter.lower())
        if len(opts) >= 2 and 0 <= ci < len(opts):
            rows.append((q, opts, ci))
    return rows


def parse_questions(html):
    rows = parse_format_a(html)
    if not rows:
        rows = parse_format_c(html)
    if not rows:
        rows = parse_format_d(html)
    return rows


def classify(slug):
    if slug in CPP_SLUGS:
        return "C++ Programming", "T4Tutorials C++ MCQs"
    if slug in CS_SUBJECTS:
        return CS_SUBJECTS[slug], "T4Tutorials " + slug.replace("-", " ").title()
    if CS_KEYWORDS.search(slug):
        return "Computer Science General Test", "T4Tutorials " + slug.replace("-", " ").title()
    return None, None


def main():
    done = set()
    if os.path.exists(PROG):
        done = set(l.strip() for l in open(PROG, encoding="utf-8") if l.strip())
    seen = {}
    if os.path.exists(OUT):
        for ln in open(OUT, encoding="utf-8"):
            try:
                j = json.loads(ln)
                seen.setdefault(norm(j["question"]), j)
            except Exception:
                pass
    fout = open(OUT, "a", encoding="utf-8")
    flog = open(LOG, "a", encoding="utf-8")
    crash = open(os.path.join(BASE, "t4_crash.log"), "a", encoding="utf-8")
    new = 0
    for slug in SLUGS:
        if slug in done:
            continue
        url = f"https://t4tutorials.com/{slug}/"
        try:
            html = get(url)
            if not html:
                flog.write(f"{slug}: NO HTML\n")
                flog.flush()
                with open(PROG, "a", encoding="utf-8") as f:
                    f.write(slug + "\n")
                continue
            subj, topic = classify(slug)
            import threading
            res = {}
            thr = threading.Thread(target=lambda: res.update(rows=parse_questions(html)))
            thr.start()
            thr.join(60)
            if thr.is_alive():
                flog.write(f"{slug}: PARSE TIMEOUT\n")
                flog.flush()
                rows = []
            else:
                rows = res.get("rows", [])
            cnt = 0
            for q, opts, ci in rows:
                k = norm(q)
                if k in seen:
                    continue
                it = {
                    "subject": subj or slug,
                    "topic": topic or slug,
                    "question": q,
                    "options": opts,
                    "correct": ci + 1,
                    "explanation": "",
                    "source_url": url,
                }
                seen[k] = it
                fout.write(json.dumps(it, ensure_ascii=False) + "\n")
                cnt += 1
                new += 1
            fout.flush()
            flog.write(f"{slug}: {len(rows)} parsed, {cnt} new\n")
            flog.flush()
            with open(PROG, "a", encoding="utf-8") as f:
                f.write(slug + "\n")
            print(f"{slug}: {len(rows)} parsed, {cnt} new (total new {new})", flush=True)
        except Exception as e:
            import traceback
            crash.write(f"CRASH {url}: {e}\n{traceback.format_exc()}\n")
            crash.flush()
            with open(PROG, "a", encoding="utf-8") as f:
                f.write(slug + "\n")
            print(f"{slug}: CRASH {e}", flush=True)
        time.sleep(0.4)
    fout.close()
    flog.close()
    print("DONE total new:", new)


if __name__ == "__main__":
    main()