File "fetch_indiabix_cpp.py"

Full path: /home/algopkco/public_html/scraper/fetch_indiabix_cpp.py
File size: 5.75 B (5.75 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8

Download   Open   Edit   Advanced Editor &nnbsp; Back

import hashlib
import json
import os
import re
import time

import requests

BASE = os.path.dirname(os.path.abspath(__file__))
DATA = os.path.join(BASE, "data")
OUT = os.path.join(DATA, "indiabix_cpp_mcqs.jsonl")
PROG = os.path.join(DATA, "ibx_cpp_progress.txt")

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/126.0 Safari/537.36"}
s = requests.Session()
s.headers.update(HEADERS)

SUBJECT = "C++ Programming"


def get(url, tries=5):
    for i in range(tries):
        try:
            r = s.get(url, timeout=30)
            if r.status_code == 200:
                return r.text
            print("  HTTP", r.status_code, url, flush=True)
            if r.status_code == 404:
                return None
        except Exception as e:
            print("  err", e, "retry", i, flush=True)
        time.sleep(1 + i * 2)
    return None


def clean_html(t):
    t = re.sub(r"<br\s*/?>", "\n", t)
    t = re.sub(r"<[^>]+>", "", t)
    t = re.sub(r"\xa0", " ", t)
    t = re.sub(r"[ \t]+", " ", t)
    t = re.sub(r"\n\s*\n+", "\n", t)
    return t.strip()


def split_blocks(html):
    idxs = [m.start() for m in re.finditer(r'<div class="bix-div-container">', html)]
    if not idxs:
        return []
    blocks = []
    for i, s in enumerate(idxs):
        e = idxs[i + 1] if i + 1 < len(idxs) else len(html)
        blocks.append(html[s:e])
    return blocks


def parse_page(html, url, topic):
    qs = []
    for b in split_blocks(html):
        head = b.split('class="jq-hdnakq"', 1)[0]
        if "<img" in head:
            continue
        qm = re.search(r'class="bix-td-qtxt[^"]*">(.*?)</div>', b, re.S)
        if not qm:
            continue
        question = clean_html(qm.group(1))
        if not question:
            continue
        options = []
        for row in re.findall(r'id="tdOptionDt_[A-Z]_[0-9]+">(.*?)\s*</div>\s*</div>', b, re.S):
            val = clean_html(row)
            if val and val not in options:
                options.append(val)
        if len(options) < 2:
            continue
        am = re.search(r'class="jq-hdnakq"[^>]*value="([A-Z])"', b)
        if not am:
            continue
        idx = "ABCDEFGHIJKLMNOPQRSTUVWXYZ".index(am.group(1))
        if idx >= len(options):
            continue
        tail = b.split("class=bix-ans-description", 1)
        if len(tail) == 1:
            tail = b.split('class="bix-ans-description', 1)
        explanation = ""
        if len(tail) > 1:
            desc = tail[1]
            end = desc.find("bix-div-workspace")
            if end != -1:
                desc = desc[:end]
            m2 = re.search(r">(.*)$", desc, re.S)
            if m2:
                explanation = clean_html(m2.group(1))
                cut = explanation.rfind("<div class=")
                if cut != -1:
                    explanation = explanation[:cut].strip()
        qs.append({
            "subject": SUBJECT,
            "topic": topic,
            "question": question,
            "options": options,
            "correct": idx + 1,
            "explanation": explanation[:2000],
            "source_url": url,
        })
    return qs


def main():
    done = set()
    if os.path.exists(PROG):
        done = {l.strip() for l in open(PROG, encoding="utf-8")}
    seen = set()
    if os.path.exists(OUT):
        for ln in open(OUT, encoding="utf-8"):
            try:
                seen.add(json.loads(ln)["question"])
            except Exception:
                pass
    chapters = []
    h = get("https://www.indiabix.com/cpp-programming/")
    if h:
        for l in sorted(set(re.findall(r'href="(https://www\.indiabix\.com/cpp-programming/[^"]+)"', h))):
            slug = l.split("/")[-2]
            if slug in ("questions-and-answers", "discussion"):
                continue
            chapters.append((l, slug.replace("-", " ").title()))
    print("chapters:", chapters, flush=True)
    fout = open(OUT, "a", encoding="utf-8")
    total = 0
    new = 0
    for url, topic in chapters:
        section = "cpp-programming"
        cat = url.split("/")[-2]
        cur = url
        pages = 0
        while True:
            if cur in done:
                nl = numbered_links_from(html, section, cat) if False else None
                break
            html = get(cur)
            if html is None:
                break
            qs = parse_page(html, cur, topic)
            for it in qs:
                if it["question"] in seen:
                    continue
                seen.add(it["question"])
                fout.write(json.dumps(it, ensure_ascii=False) + "\n")
                new += 1
            total += len(qs)
            pages += 1
            nl = re.findall(r'href="(https://www\.indiabix\.com/' + section + r'/' + cat + r'/[0-9]{4,})"[^>]*>(\d+)', html)
            if not nl:
                break
            m = re.search(r'<li class="page-item active"[^>]*>.*?<span class="page-link">(\d+)</span>', html, re.S)
            cur_pg = int(m.group(1)) if m else 1
            nxt = None
            for href, pg in nl:
                if int(pg) == cur_pg + 1:
                    nxt = href
                    break
            if nxt is None:
                cand = [h for h, p in nl if int(p) > cur_pg]
                if cand:
                    nxt = cand[0]
            if nxt is None:
                break
            with open(PROG, "a", encoding="utf-8") as pf:
                pf.write(cur + "\n")
            cur = nxt
            fout.flush()
            print(f"{topic} p{pages}: {len(qs)} q (new {new})", flush=True)
            time.sleep(0.8)
        with open(PROG, "a", encoding="utf-8") as pf:
            pf.write(url + "\n")
        print(f"{topic}: DONE ({pages} pages)", flush=True)
    fout.close()
    print("DONE total:", total, "new:", new)


if __name__ == "__main__":
    main()