File "fetch_tpointtech.py"

Full path: /home/algopkco/public_html/scraper/fetch_tpointtech.py
File size: 11.3 B (11.3 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8

Download   Open   Edit   Advanced Editor &nnbsp; Back

import html as H
import json
import os
import re
import time

import requests

BASE = os.path.dirname(os.path.abspath(__file__))
DATA = os.path.join(BASE, "data")
MENU_FILE = os.path.join(DATA, "tpointtech_menu.txt")
OUT_FILE = os.path.join(DATA, "tpointtech_mcqs.jsonl")
PROGRESS_FILE = os.path.join(DATA, "tpointtech_progress.txt")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36"
}

SUBJECTS = {
    "cpp-mcq": "C++ Programming",
    "c-language-mcq": "C Programming",
    "python-mcq": "Python Programming",
    "java-mcq": "Java Programming",
    "java-multithreading-mcqs": "Java Programming",
    "thread-priority-mcqs-in-java": "Java Programming",
    "javascript-mcq": "JavaScript",
    "jquery-mcq": "jQuery",
    "dbms-mcq": "Database Management System",
    "mcqs-of-entity-relationship-diagram": "Database Management System",
    "mcqs-on-normalization": "Database Management System",
    "mcqs-on-relational-algebra": "Database Management System",
    "mcqs-on-relational-calculus": "Database Management System",
    "mcqs-on-transactions-and-concurrency-control": "Database Management System",
    "sql-mcq": "SQL",
    "html-mcq": "HTML",
    "css-mcq": "CSS",
    "bootstrap-mcq": "CSS",
    "bootstrap-4-mcq": "CSS",
    "data-structure-mcq": "Data Structures And Algorithms",
    "mcqs-on-sorting-techniques": "Data Structures And Algorithms",
    "mcqs-on-splay-tree": "Data Structures And Algorithms",
    "mcqs-on-hamiltonian-graph": "Data Structures And Algorithms",
    "mcqs-on-knapsack-algorithm": "Data Structures And Algorithms",
    "mcq-on-divide-and-conquer-algorithm": "Data Structures And Algorithms",
    "mcqs-on-greedy-algorithm": "Data Structures And Algorithms",
    "mcqs-on-n-queens-problem": "Data Structures And Algorithms",
    "mcqs-on-topological-sorting": "Data Structures And Algorithms",
    "operating-system-mcq": "Operating System",
    "computer-network-mcq": "Computer Networks",
    "oops-mcq": "Object Oriented Programming",
    "digital-electronics-mcq": "Digital Electronics",
    "compiler-design-mcq": "Compiler Design",
    "software-engineering-mcq": "Software Engineering",
    "software-testing-mcq": "Software Engineering",
    "computer-fundamental-mcq": "Computer Fundamentals",
    "computer-awareness-mcq-for-bank-exam": "Computer Fundamentals",
    "mcq-on-computer-hardware-for-bank-exam": "Computer Fundamentals",
    "mcq-on-computer-software-for-bank-exams": "Computer Fundamentals",
    "artificial-intelligence-mcq": "Artificial Intelligence",
    "data-mining-mcq": "Data Mining",
    "cloud-computing-mcq": "Cloud Computing",
    "iot-mcq": "IoT",
    "cyber-security-mcq": "Cyber Security",
    "mobile-computing-mcq": "Mobile Computing",
    "computer-architecture-mcq": "Computer Architecture",
    "computer-graphics-mcq": "Computer Graphics",
    "digital-image-processing-mcq": "Digital Image Processing",
    "digital-communication-mcq": "Electronics Engineering",
    "digital-signal-processing-mcq": "Electronics Engineering",
    "soft-computing-mcq": "Soft Computing",
    "embedded-systems-mcq": "Embedded Systems",
    "electrical-mcq": "Electrical Engineering",
    "transformer-mcq": "Electrical Engineering",
    "control-system-mcq": "Electrical Engineering",
    "power-electronics-mcq": "Electrical Engineering",
    "mechanical-engineering-mcq": "Mechanical Engineering",
    "fluid-mechanics-mcq": "Mechanical Engineering",
    "engineering-mechanics-mcq": "Mechanical Engineering",
    "civil-engineering-mcq": "Civil Engineering",
    "genetics-algorithm-mcq": "Genetic Algorithms",
    "discrete-mathematics-mcq": "Mathematics",
    "probability-mcq": "Mathematics",
    "statistics-mcq": "Mathematics",
    "number-system-mcq": "Mathematics",
    "trigonometry-mcq": "Mathematics",
    "reasoning-mcq": "Reasoning",
    "gk-mcq": "General Knowledge",
    "general-science-mcq": "General Knowledge",
    "ancient-history-mcq": "General Knowledge",
    "indian-constitution-mcq": "General Knowledge",
    "human-rights-mcq": "General Knowledge",
    "mcq-of-important-discoveries-and-invention": "General Knowledge",
    "mcq-on-vitamins-and-nutrition": "General Knowledge",
    "environmental-science-mcq": "Environmental Science",
    "environmental-studies-mcq": "Environmental Science",
    "life-processes-mcq": "School Science",
    "class-10th-science-mcq": "School Science",
    "class-9th-science-mcq": "School Science",
    "class-12-physics-mcq": "School Science",
    "biotechnology-mcq": "Biotechnology",
    "psychology-mcq": "Psychology",
    "human-resource-management-mcq": "Management",
    "research-methodology-mcq": "Management",
    "english-grammar-mcq-for-competitive-exam": "English Grammar",
    "powerpoint-mcq": "Microsoft Office",
    "digital-marketing-mcq": "Digital Marketing",
}

SKIP_PAGES = {"mcqs-preparation"}

LETTERS = "ABCDEFGHIJKLMNOPQRSTUVWXYZ"

session = requests.Session()
session.headers.update(HEADERS)


def get(url, tries=5):
    for i in range(tries):
        try:
            r = session.get(url, timeout=30)
            if r.status_code == 200:
                return r.text
            print("  HTTP", r.status_code, url, flush=True)
            if r.status_code == 404:
                return None
        except Exception as e:
            print("  err", e, "retry", i, flush=True)
        time.sleep(1 + i * 2)
    return None


def clean(s):
    s = re.sub(r"<br\s*/?>", " ", s)
    s = re.sub(r"<[^>]+>", "", s)
    s = H.unescape(s)
    s = re.sub(r"\xa0", " ", s)
    s = re.sub(r"[ \t]+", " ", s)
    s = re.sub(r"\n\s*\n+", "\n", s)
    return s.strip()


PQ = re.compile(r'<p\s+class=["\']?pq["\']?>\s*', re.I)


def parse_page(html, page_url, subject, topic):
    chunks = PQ.split(html)
    out = []
    for chunk in chunks[1:]:
        ocut = chunk.find("<ol class=")
        if ocut == -1:
            ocut = chunk.find('<ol class="')
        qpart = chunk if ocut == -1 else chunk[:ocut]
        om = re.search(r'<ol\s+class=["\']?pointsa["\']?>(.*?)</ol>', chunk, re.S)
        if om:
            options = [clean(o) for o in re.findall(r"<li>(.*?)</li>", om.group(1), re.S)]
            options = [o for o in options if o]
            qtext = re.sub(r"^\d+\)?\.?\s*", "", clean(re.sub(r"</p>", " ", qpart))).strip()
        else:
            parts = re.split(r"<p>\s*([a-zA-Z])\.\s*([^<]*)</p>", qpart)
            options = []
            for pi in range(1, len(parts), 3):
                lbl, inline, blk = parts[pi], parts[pi + 1], parts[pi + 2]
                txt = ""
                if inline.strip():
                    txt = clean(inline)
                else:
                    tas = re.findall(r"<textarea[^>]*>(.*?)</textarea>", blk, re.S)
                    if tas:
                        txt = " / ".join(H.unescape(re.sub(r"<[^>]+>", "", t)).strip() for t in tas if t.strip())
                    else:
                        txt = clean(blk)
                if txt:
                    options.append(txt)
            qtext = re.sub(r"^\d+\)?\.?\s*", "", clean(re.sub(r"</p>", " ", parts[0]))).strip()
        if not qtext:
            continue
        cm = re.search(r"<textarea[^>]*>(.*?)</textarea>", qpart, re.S) or re.search(
            r"<pre[^>]*>(.*?)</pre>", qpart, re.S
        )
        if cm:
            code = H.unescape(re.sub(r"<[^>]+>", "", cm.group(1))).strip()
            if code:
                qtext = qtext + "\n\n```\n" + code + "\n```"
        if len(options) < 2:
            continue
        am = re.search(r'<div\s+class=["\']?testanswer["\']?[^>]*>(.*?)</div>', chunk, re.S)
        if not am:
            continue
        amt = am.group(1)
        amatch = re.search(r"Answer:\s*(?:<[^>]+>\s*)*([A-Z])\b", amt, re.I)
        if not amatch:
            amatch = re.search(r"Explanation:\s*(?:<[^>]+>\s*)*([A-Z])\b", amt, re.I)
        if not amatch:
            continue
        ci = LETTERS.upper().find(amatch.group(1).upper())
        if ci < 0 or ci >= len(options):
            continue
        emi = amt.find("Explanation:")
        expl = clean(amt[emi + len("Explanation:"):]) if emi != -1 else ""
        out.append(
            {
                "subject": subject,
                "topic": topic,
                "question": qtext,
                "options": options,
                "correct": ci + 1,
                "explanation": expl,
                "source_url": page_url,
            }
        )
    return out


def topic_name(title, page_url):
    t = H.unescape(re.sub(r"<[^>]+>", "", title or "")).strip()
    t = re.sub(r"MCQ'?s|MCQs|MCQ", "", t, flags=re.I)
    t = re.sub(r"\s*Part\s*\d+$", "", t, flags=re.I)
    t = re.sub(r"\s+", " ", t).strip()
    return t or page_url


def main():
    menu = []
    if os.path.exists(MENU_FILE):
        with open(MENU_FILE, encoding="utf-8") as f:
            for ln in f:
                parts = ln.rstrip("\n").split("\t")
                if len(parts) == 2:
                    menu.append((parts[0], parts[1]))
    else:
        print("menu file missing, crawling cpp-mcq only")
        menu = [("https://www.tpointtech.com/cpp-mcq", "C++ MCQ")]
    done = set()
    if os.path.exists(PROGRESS_FILE):
        with open(PROGRESS_FILE, encoding="utf-8") as f:
            done = {ln.strip() for ln in f}
    fout = open(OUT_FILE, "a", encoding="utf-8")
    total = 0
    seen_questions = set()
    if os.path.exists(OUT_FILE):
        with open(OUT_FILE, encoding="utf-8") as f:
            for ln in f:
                try:
                    it = json.loads(ln)
                    seen_questions.add(it["question"])
                except Exception:
                    pass
    pages = []
    for url, title in menu:
        m = re.search(r"tpointtech\.com/([^/]+)", url)
        if not m or m.group(1) in SKIP_PAGES:
            continue
        base = m.group(1)
        subj = None
        for k, v in SUBJECTS.items():
            if base == k or base.startswith(k + "-part"):
                subj = v
                break
        if not subj:
            continue
        pages.append((url, base, title, subj))
        for i in range(2, 10):
            variant_a = f"{base}-part-{i}"
            variant_b = f"{base}-part{i}"
            for v in (variant_a, variant_b):
                if v == base:
                    continue
                pages.append((f"https://www.tpointtech.com/{v}", v, f"{title} Part {i}", subj))
    print("pages to crawl:", len(pages), flush=True)
    new = 0
    for url, slug, title, subj in pages:
        if url in done:
            continue
        try:
            html = get(url)
            if not html:
                continue
            qs = parse_page(html, url, subj, topic_name(title, slug))
            for it in qs:
                if it["question"] in seen_questions:
                    continue
                seen_questions.add(it["question"])
                fout.write(json.dumps(it, ensure_ascii=False) + "\n")
                new += 1
            total += len(qs)
        except Exception as e:
            print("ERR on", slug, e, flush=True)
            continue
        with open(PROGRESS_FILE, "a", encoding="utf-8") as pf:
            pf.write(url + "\n")
        fout.flush()
        print(f"{slug}: {len(qs)} q (new {new}, total {total})", flush=True)
        time.sleep(0.6)
    fout.close()
    print("DONE pages done:", len(done), "questions total:", total, "new:", new)


if __name__ == "__main__":
    main()