File "fetch_mcqsets_more.py"
Full path: /home/algopkco/public_html/scraper/fetch_mcqsets_more.py
File
size: 5.18 B (5.18 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8
Download Open Edit Advanced Editor &nnbsp; Back
import json
import os
import re
import time
import requests
BASE = os.path.dirname(os.path.abspath(__file__))
DATA = os.path.join(BASE, "data")
OUT = os.path.join(DATA, "mcqsets_more_mcqs.jsonl")
PROG = os.path.join(BASE, "mcqsets_more_progress.txt")
LOG = os.path.join(BASE, "mcqsets_more_run.log")
H = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/126.0 Safari/537.36"}
s = requests.Session()
s.headers.update(H)
# prefix, subject, topic
CATS = [
("mcq-questions-2/c", "C Programming", "mcqsets C Sets"),
("mcq-questions-2/data-structures-and-algorithms", "Data Structures And Algorithms", "mcqsets DSA"),
("mcq-questions-2/computer-fundamentals", "Computer Science General Test", "mcqsets Fundamentals"),
("mcq-questions-2/operating-system", "Operating Systems", "mcqsets OS"),
("mcq-questions-2/html-web-page-designing", "Web Technologies", "mcqsets HTML"),
("mcq-questions-2/microsoft-word", "Computer Science General Test", "mcqsets MS Word"),
("mcq-questions-2/ms-excel", "Computer Science General Test", "mcqsets MS Excel"),
("mcq-questions-2/ms-powerpoint", "Computer Science General Test", "mcqsets MS PowerPoint"),
("mcq-questions-2/ms-access-mcq-questions-2", "Database Management Systems", "mcqsets MS Access"),
("mcq-questions-collection", "Computer Science General Test", "mcqsets Collection"),
]
FORMAT_LOWER = re.compile(
r"(\d+)\.\s+(.{15,}?)\s+a\.\s+(.{1,300}?)\s+b\.\s+(.{1,300}?)\s+c\.\s+(.{1,300}?)\s+d\.\s+(.{1,300}?)(?=\s+\d+\.|Answers? to|Filed Under| |$)",
re.S,
)
FORMAT_UPPER = re.compile(
r"(\d+)\.\s+(.{15,}?)\s+A\)\s+(.{1,300}?)\s+B\)\s+(.{1,300}?)\s+C\)\s+(.{1,300}?)\s+D\)\s+(.{1,300}?)(?=\s+\d+\.|Answers? to|Filed Under| |$)",
re.S,
)
ANS_KEY = re.compile(r"Answers? to[^0-9]{0,120}?((?:\d+\s*[\-\u2011\u2013\u2014]?\s*[a-eA-E]\s*)+)")
def get(url, tries=3):
for i in range(tries):
try:
r = s.get(url, timeout=30)
if r.status_code == 200:
return r.text
except Exception:
pass
time.sleep(1 + i)
return None
def parse(html):
body = re.sub(r"<script.*?</script>|<style.*?</style>", "", html, flags=re.S)
plain = re.sub(r"<[^>]+>", " ", body)
plain = re.sub(r"\s+", " ", plain)
plain = re.sub(r"&#?\w+;", " ", plain)
qs = []
for m in FORMAT_LOWER.finditer(plain):
qs.append((int(m.group(1)), m.group(2).strip(), [m.group(3).strip(), m.group(4).strip(), m.group(5).strip(), m.group(6).strip()]))
if not qs:
for m in FORMAT_UPPER.finditer(plain):
qs.append((int(m.group(1)), m.group(2).strip(), [m.group(3).strip(), m.group(4).strip(), m.group(5).strip(), m.group(6).strip()]))
ans = {}
am = ANS_KEY.search(plain)
if am:
for pair in re.findall(r"(\d+)\s*[\-\u2011\u2013\u2014]?\s*([a-eA-E])", am.group(1)):
ans[int(pair[0])] = pair[1].lower()
rows = []
for num, q, opts in qs:
let = ans.get(num)
if not let:
continue
ci = "abcde".find(let)
if ci < 0 or ci >= len(opts):
continue
if len(q) < 10:
continue
rows.append((q, opts, ci))
return rows
def main():
done = set()
if os.path.exists(PROG):
done = set(l.strip() for l in open(PROG, encoding="utf-8") if l.strip())
seen = set()
if os.path.exists(OUT):
for ln in open(OUT, encoding="utf-8"):
try:
seen.add(json.loads(ln)["question"])
except Exception:
pass
fout = open(OUT, "a", encoding="utf-8")
flog = open(LOG, "a", encoding="utf-8")
new = 0
for path, subj, topic in CATS:
url = f"https://mcqsets.com/{path}/"
html = get(url)
if not html:
flog.write(f"{path}: NO HTML\n")
flog.flush()
continue
slugs = sorted(set(re.findall(r'href="/(?:' + re.escape(path) + r')/([a-z0-9\-]+)/?"', html)))
queue = [f"/{path}/" + sl + "/" for sl in slugs]
queue.insert(0, url)
for u in queue:
if u in done:
continue
h = get(u)
if not h:
flog.write(f"{u}: NO HTML\n")
flog.flush()
with open(PROG, "a", encoding="utf-8") as f:
f.write(u + "\n")
continue
rows = parse(h)
cnt = 0
for q, opts, ci in rows:
if q in seen:
continue
seen.add(q)
fout.write(json.dumps({
"subject": subj, "topic": topic, "question": q,
"options": opts, "correct": ci + 1, "explanation": "", "source_url": u,
}, ensure_ascii=False) + "\n")
cnt += 1
new += 1
fout.flush()
print(f"{u}: {len(rows)} parsed, {cnt} new (total {new})", flush=True)
flog.write(f"{u}: {len(rows)} parsed, {cnt} new\n")
flog.flush()
with open(PROG, "a", encoding="utf-8") as f:
f.write(u + "\n")
time.sleep(0.5)
fout.close()
flog.close()
print("DONE total new:", new)
if __name__ == "__main__":
main()