File "fetch_mcqlearn.py"
Full path: /home/algopkco/public_html/scraper/fetch_mcqlearn.py
File
size: 5.94 B (5.94 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8
Download Open Edit Advanced Editor &nnbsp; Back
import json
import re
import html
import os
import time
import threading
from concurrent.futures import ThreadPoolExecutor, as_completed
import requests
H = {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36',
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
}
HOME = 'https://mcqlearn.com/'
DATA = r'D:\xampp\htdocs\algo\scraper\data\mcqlearn_mcqs.jsonl'
PROG = r'D:\xampp\htdocs\algo\scraper\mcqlearn_progress.txt'
MAXPAGE = 60
NTHREADS = 10
SUBJ_MAP = [
(re.compile(r'(?i)/grade10/physics|/grade9/physics|/grade8/physics|/physics/'), 'Physics Mcqs'),
(re.compile(r'(?i)/grade10/chemistry|/grade9/chemistry|/chemistry/'), 'Chemistry Mcqs'),
(re.compile(r'(?i)/grade10/biology|/grade9/biology|/biology/'), 'Biology Mcqs'),
(re.compile(r'(?i)/math'), 'Mathematics'),
(re.compile(r'(?i)/english'), 'English Mcqs'),
(re.compile(r'(?i)/computer|/it-|data'), 'Computer Science General Test'),
(re.compile(r'(?i)/geography|/earth'), 'Geography'),
(re.compile(r'(?i)/science|inventions'), 'School Science'),
]
RE_ALPHANUM = re.compile(r'[^a-z0-9]')
def subj_of(url):
u = url.split('?')[0]
for pat, s in SUBJ_MAP:
if pat.search(u):
return s
return 'School Science'
def collect_links(html_text, pattern=None):
out = set()
for m in re.finditer(r'href="(https?://mcqlearn\.com/[^"]+?\.php[^"]*)"', html_text):
u = html.unescape(m.group(1)).split('#')[0]
u = re.sub(r'\?page=\d+$', '', u)
if pattern and not pattern.search(u):
continue
out.add(u)
return sorted(out)
HW = (re.compile(r'(?i)(quiz-questions-and-answers|mcqs-and-quizzes|exam-questions|exam-tests|multiple-choice-questions)'),)
MAXHUBS = 400
def is_hub(u):
return bool(HW[0].search(u))
def parse_page(h):
rows = []
for m in re.finditer(
r'<span class="mcq-formatting">MCQ\s*\d+:.*?</span>\s*(.*?)</p>'
r'\s*<ol[^>]*>(.*?)</ol>'
r'\s*<div id="answer-\d+"[^>]*>\s*([a-d])\s*</div>', h, re.S | re.I):
q = re.sub(r'<[^>]+>', '', m.group(1))
q = html.unescape(q).strip()
opts = re.findall(r'<li[^>]*id="([a-d])-\d+"[^>]*>(.*?)</li>', m.group(2), re.S | re.I)
d = {}
for letter, o in opts:
ot = html.unescape(re.sub(r'<[^>]+>', '', o)).strip()
if ot:
d[letter.lower()] = ot
if len(d) < 2 or not q:
continue
corr = m.group(3).lower()
if corr not in d:
continue
letters = ['a', 'b', 'c', 'd']
cands = [d.get(l, '') for l in letters]
if not all(cands):
continue
rows.append({
'question': q,
'options': [x if x else '' for x in cands],
'correct': letters.index(corr) + 1,
})
return rows
def has_next(h):
if re.search(r"href=[\"'][^\"']*page=\d+[\"'][^>]*>\s*Next\s*<", h, re.I):
return True
nums = [int(x) for x in re.findall(r'>(\d+)</a>', h)]
return bool(nums and max(nums) > 1)
def fetch_chapter(ch_url):
rows = []
error = None
try:
r = requests.get(ch_url, headers=H, timeout=40)
if r.status_code != 200:
return [], 'http' + str(r.status_code)
rows += parse_page(r.text)
pg = 2
while has_next(r.text) and pg <= MAXPAGE:
r = requests.get(ch_url + ('&' if '?' in ch_url else '?') + f'page={pg}', headers=H, timeout=40)
if r.status_code != 200:
break
got = parse_page(r.text)
if not got:
break
rows += got
pg += 1
time.sleep(0.15)
except Exception as e:
error = str(e)[:120]
return rows, error
def main():
home_html = requests.get(HOME, headers=H, timeout=40).text
hubs = [u for u in collect_links(home_html) if is_hub(u)]
chapters = set()
for hb in hubs[:MAXHUBS]:
try:
h = requests.get(hb, headers=H, timeout=40).text
for u in collect_links(h):
if not is_hub(u) and '?' not in u and '.php' in u:
chapters.add(u)
except Exception:
continue
time.sleep(0.05)
print('hubs:', len(hubs), 'chapter urls:', len(chapters), flush=True)
done = set()
if os.path.exists(PROG):
for ln in open(PROG, encoding='utf-8'):
done.add(ln.strip())
todo = [c for c in sorted(chapters) if c not in done]
print('todo:', len(todo), flush=True)
if not todo:
return
lock = threading.Lock()
fw = open(DATA, 'a', encoding='utf-8')
pg = open(PROG, 'a', encoding='utf-8')
stats = {'rows': 0, 'ok': 0, 'err': 0}
t0 = time.time()
with ThreadPoolExecutor(max_workers=NTHREADS) as ex:
futs = {ex.submit(fetch_chapter, c): c for c in todo}
for i, f in enumerate(as_completed(futs), 1):
ch = futs[f]
rows, err = f.result()
with lock:
subj = subj_of(ch)
topic = ch.rsplit('/', 1)[-1].split('.')[0].replace('-', ' ').title()
for rw in rows:
fw.write(json.dumps({
'subject': subj, 'question': rw['question'], 'options': rw['options'],
'correct': rw['correct'], 'topic': topic, 'source': 'mcqlearn',
'url': ch, 'explanation': '',
}, ensure_ascii=False) + '\n')
if err:
stats['err'] += 1
else:
stats['ok'] += 1
stats['rows'] += len(rows)
pg.write(ch + '\n')
if i % 20 == 0:
print(f'[{i}/{len(todo)}] rows={stats["rows"]} ok={stats["ok"]} err={stats["err"]} elapsed={time.time()-t0:.0f}s', flush=True)
fw.close()
pg.close()
print('DONE', stats, flush=True)
if __name__ == '__main__':
main()