Create New Item
×
Item Type
File
Folder
Item Name
×
Search file in folder and subfolders...
File Manager
/
scraper
Advanced Search
Upload
New Item
Settings
Back
Back Up
Advanced Editor
Save
import hashlib import json import os import re import time import requests BASE = os.path.dirname(os.path.abspath(__file__)) DATA = os.path.join(BASE, "data") OUT = os.path.join(DATA, "indiabix_cpp_mcqs.jsonl") PROG = os.path.join(DATA, "ibx_cpp_progress.txt") HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/126.0 Safari/537.36"} s = requests.Session() s.headers.update(HEADERS) SUBJECT = "C++ Programming" def get(url, tries=5): for i in range(tries): try: r = s.get(url, timeout=30) if r.status_code == 200: return r.text print(" HTTP", r.status_code, url, flush=True) if r.status_code == 404: return None except Exception as e: print(" err", e, "retry", i, flush=True) time.sleep(1 + i * 2) return None def clean_html(t): t = re.sub(r"<br\s*/?>", "\n", t) t = re.sub(r"<[^>]+>", "", t) t = re.sub(r"\xa0", " ", t) t = re.sub(r"[ \t]+", " ", t) t = re.sub(r"\n\s*\n+", "\n", t) return t.strip() def split_blocks(html): idxs = [m.start() for m in re.finditer(r'<div class="bix-div-container">', html)] if not idxs: return [] blocks = [] for i, s in enumerate(idxs): e = idxs[i + 1] if i + 1 < len(idxs) else len(html) blocks.append(html[s:e]) return blocks def parse_page(html, url, topic): qs = [] for b in split_blocks(html): head = b.split('class="jq-hdnakq"', 1)[0] if "<img" in head: continue qm = re.search(r'class="bix-td-qtxt[^"]*">(.*?)</div>', b, re.S) if not qm: continue question = clean_html(qm.group(1)) if not question: continue options = [] for row in re.findall(r'id="tdOptionDt_[A-Z]_[0-9]+">(.*?)\s*</div>\s*</div>', b, re.S): val = clean_html(row) if val and val not in options: options.append(val) if len(options) < 2: continue am = re.search(r'class="jq-hdnakq"[^>]*value="([A-Z])"', b) if not am: continue idx = "ABCDEFGHIJKLMNOPQRSTUVWXYZ".index(am.group(1)) if idx >= len(options): continue tail = b.split("class=bix-ans-description", 1) if len(tail) == 1: tail = b.split('class="bix-ans-description', 1) explanation = "" if len(tail) > 1: desc = tail[1] end = desc.find("bix-div-workspace") if end != -1: desc = desc[:end] m2 = re.search(r">(.*)$", desc, re.S) if m2: explanation = clean_html(m2.group(1)) cut = explanation.rfind("<div class=") if cut != -1: explanation = explanation[:cut].strip() qs.append({ "subject": SUBJECT, "topic": topic, "question": question, "options": options, "correct": idx + 1, "explanation": explanation[:2000], "source_url": url, }) return qs def main(): done = set() if os.path.exists(PROG): done = {l.strip() for l in open(PROG, encoding="utf-8")} seen = set() if os.path.exists(OUT): for ln in open(OUT, encoding="utf-8"): try: seen.add(json.loads(ln)["question"]) except Exception: pass chapters = [] h = get("https://www.indiabix.com/cpp-programming/") if h: for l in sorted(set(re.findall(r'href="(https://www\.indiabix\.com/cpp-programming/[^"]+)"', h))): slug = l.split("/")[-2] if slug in ("questions-and-answers", "discussion"): continue chapters.append((l, slug.replace("-", " ").title())) print("chapters:", chapters, flush=True) fout = open(OUT, "a", encoding="utf-8") total = 0 new = 0 for url, topic in chapters: section = "cpp-programming" cat = url.split("/")[-2] cur = url pages = 0 while True: if cur in done: nl = numbered_links_from(html, section, cat) if False else None break html = get(cur) if html is None: break qs = parse_page(html, cur, topic) for it in qs: if it["question"] in seen: continue seen.add(it["question"]) fout.write(json.dumps(it, ensure_ascii=False) + "\n") new += 1 total += len(qs) pages += 1 nl = re.findall(r'href="(https://www\.indiabix\.com/' + section + r'/' + cat + r'/[0-9]{4,})"[^>]*>(\d+)', html) if not nl: break m = re.search(r'<li class="page-item active"[^>]*>.*?<span class="page-link">(\d+)</span>', html, re.S) cur_pg = int(m.group(1)) if m else 1 nxt = None for href, pg in nl: if int(pg) == cur_pg + 1: nxt = href break if nxt is None: cand = [h for h, p in nl if int(p) > cur_pg] if cand: nxt = cand[0] if nxt is None: break with open(PROG, "a", encoding="utf-8") as pf: pf.write(cur + "\n") cur = nxt fout.flush() print(f"{topic} p{pages}: {len(qs)} q (new {new})", flush=True) time.sleep(0.8) with open(PROG, "a", encoding="utf-8") as pf: pf.write(url + "\n") print(f"{topic}: DONE ({pages} pages)", flush=True) fout.close() print("DONE total:", total, "new:", new) if __name__ == "__main__": main()