File "fetch_indiabix_cpp.py"
Full path: /home/algopkco/public_html/scraper/fetch_indiabix_cpp.py
File
size: 5.75 B (5.75 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8
Download Open Edit Advanced Editor &nnbsp; Back
import hashlib
import json
import os
import re
import time
import requests
BASE = os.path.dirname(os.path.abspath(__file__))
DATA = os.path.join(BASE, "data")
OUT = os.path.join(DATA, "indiabix_cpp_mcqs.jsonl")
PROG = os.path.join(DATA, "ibx_cpp_progress.txt")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/126.0 Safari/537.36"}
s = requests.Session()
s.headers.update(HEADERS)
SUBJECT = "C++ Programming"
def get(url, tries=5):
for i in range(tries):
try:
r = s.get(url, timeout=30)
if r.status_code == 200:
return r.text
print(" HTTP", r.status_code, url, flush=True)
if r.status_code == 404:
return None
except Exception as e:
print(" err", e, "retry", i, flush=True)
time.sleep(1 + i * 2)
return None
def clean_html(t):
t = re.sub(r"<br\s*/?>", "\n", t)
t = re.sub(r"<[^>]+>", "", t)
t = re.sub(r"\xa0", " ", t)
t = re.sub(r"[ \t]+", " ", t)
t = re.sub(r"\n\s*\n+", "\n", t)
return t.strip()
def split_blocks(html):
idxs = [m.start() for m in re.finditer(r'<div class="bix-div-container">', html)]
if not idxs:
return []
blocks = []
for i, s in enumerate(idxs):
e = idxs[i + 1] if i + 1 < len(idxs) else len(html)
blocks.append(html[s:e])
return blocks
def parse_page(html, url, topic):
qs = []
for b in split_blocks(html):
head = b.split('class="jq-hdnakq"', 1)[0]
if "<img" in head:
continue
qm = re.search(r'class="bix-td-qtxt[^"]*">(.*?)</div>', b, re.S)
if not qm:
continue
question = clean_html(qm.group(1))
if not question:
continue
options = []
for row in re.findall(r'id="tdOptionDt_[A-Z]_[0-9]+">(.*?)\s*</div>\s*</div>', b, re.S):
val = clean_html(row)
if val and val not in options:
options.append(val)
if len(options) < 2:
continue
am = re.search(r'class="jq-hdnakq"[^>]*value="([A-Z])"', b)
if not am:
continue
idx = "ABCDEFGHIJKLMNOPQRSTUVWXYZ".index(am.group(1))
if idx >= len(options):
continue
tail = b.split("class=bix-ans-description", 1)
if len(tail) == 1:
tail = b.split('class="bix-ans-description', 1)
explanation = ""
if len(tail) > 1:
desc = tail[1]
end = desc.find("bix-div-workspace")
if end != -1:
desc = desc[:end]
m2 = re.search(r">(.*)$", desc, re.S)
if m2:
explanation = clean_html(m2.group(1))
cut = explanation.rfind("<div class=")
if cut != -1:
explanation = explanation[:cut].strip()
qs.append({
"subject": SUBJECT,
"topic": topic,
"question": question,
"options": options,
"correct": idx + 1,
"explanation": explanation[:2000],
"source_url": url,
})
return qs
def main():
done = set()
if os.path.exists(PROG):
done = {l.strip() for l in open(PROG, encoding="utf-8")}
seen = set()
if os.path.exists(OUT):
for ln in open(OUT, encoding="utf-8"):
try:
seen.add(json.loads(ln)["question"])
except Exception:
pass
chapters = []
h = get("https://www.indiabix.com/cpp-programming/")
if h:
for l in sorted(set(re.findall(r'href="(https://www\.indiabix\.com/cpp-programming/[^"]+)"', h))):
slug = l.split("/")[-2]
if slug in ("questions-and-answers", "discussion"):
continue
chapters.append((l, slug.replace("-", " ").title()))
print("chapters:", chapters, flush=True)
fout = open(OUT, "a", encoding="utf-8")
total = 0
new = 0
for url, topic in chapters:
section = "cpp-programming"
cat = url.split("/")[-2]
cur = url
pages = 0
while True:
if cur in done:
nl = numbered_links_from(html, section, cat) if False else None
break
html = get(cur)
if html is None:
break
qs = parse_page(html, cur, topic)
for it in qs:
if it["question"] in seen:
continue
seen.add(it["question"])
fout.write(json.dumps(it, ensure_ascii=False) + "\n")
new += 1
total += len(qs)
pages += 1
nl = re.findall(r'href="(https://www\.indiabix\.com/' + section + r'/' + cat + r'/[0-9]{4,})"[^>]*>(\d+)', html)
if not nl:
break
m = re.search(r'<li class="page-item active"[^>]*>.*?<span class="page-link">(\d+)</span>', html, re.S)
cur_pg = int(m.group(1)) if m else 1
nxt = None
for href, pg in nl:
if int(pg) == cur_pg + 1:
nxt = href
break
if nxt is None:
cand = [h for h, p in nl if int(p) > cur_pg]
if cand:
nxt = cand[0]
if nxt is None:
break
with open(PROG, "a", encoding="utf-8") as pf:
pf.write(cur + "\n")
cur = nxt
fout.flush()
print(f"{topic} p{pages}: {len(qs)} q (new {new})", flush=True)
time.sleep(0.8)
with open(PROG, "a", encoding="utf-8") as pf:
pf.write(url + "\n")
print(f"{topic}: DONE ({pages} pages)", flush=True)
fout.close()
print("DONE total:", total, "new:", new)
if __name__ == "__main__":
main()