File "fetch_mcqlearn.py"

Full path: /home/algopkco/public_html/scraper/fetch_mcqlearn.py
File size: 5.94 B (5.94 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8

Download   Open   Edit   Advanced Editor &nnbsp; Back

import json
import re
import html
import os
import time
import threading
from concurrent.futures import ThreadPoolExecutor, as_completed

import requests

H = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
}
HOME = 'https://mcqlearn.com/'
DATA = r'D:\xampp\htdocs\algo\scraper\data\mcqlearn_mcqs.jsonl'
PROG = r'D:\xampp\htdocs\algo\scraper\mcqlearn_progress.txt'
MAXPAGE = 60
NTHREADS = 10

SUBJ_MAP = [
    (re.compile(r'(?i)/grade10/physics|/grade9/physics|/grade8/physics|/physics/'), 'Physics Mcqs'),
    (re.compile(r'(?i)/grade10/chemistry|/grade9/chemistry|/chemistry/'), 'Chemistry Mcqs'),
    (re.compile(r'(?i)/grade10/biology|/grade9/biology|/biology/'), 'Biology Mcqs'),
    (re.compile(r'(?i)/math'), 'Mathematics'),
    (re.compile(r'(?i)/english'), 'English Mcqs'),
    (re.compile(r'(?i)/computer|/it-|data'), 'Computer Science General Test'),
    (re.compile(r'(?i)/geography|/earth'), 'Geography'),
    (re.compile(r'(?i)/science|inventions'), 'School Science'),
]
RE_ALPHANUM = re.compile(r'[^a-z0-9]')


def subj_of(url):
    u = url.split('?')[0]
    for pat, s in SUBJ_MAP:
        if pat.search(u):
            return s
    return 'School Science'


def collect_links(html_text, pattern=None):
    out = set()
    for m in re.finditer(r'href="(https?://mcqlearn\.com/[^"]+?\.php[^"]*)"', html_text):
        u = html.unescape(m.group(1)).split('#')[0]
        u = re.sub(r'\?page=\d+$', '', u)
        if pattern and not pattern.search(u):
            continue
        out.add(u)
    return sorted(out)


HW = (re.compile(r'(?i)(quiz-questions-and-answers|mcqs-and-quizzes|exam-questions|exam-tests|multiple-choice-questions)'),)
MAXHUBS = 400


def is_hub(u):
    return bool(HW[0].search(u))


def parse_page(h):
    rows = []
    for m in re.finditer(
            r'<span class="mcq-formatting">MCQ\s*\d+:.*?</span>\s*(.*?)</p>'
            r'\s*<ol[^>]*>(.*?)</ol>'
            r'\s*<div id="answer-\d+"[^>]*>\s*([a-d])\s*</div>', h, re.S | re.I):
        q = re.sub(r'<[^>]+>', '', m.group(1))
        q = html.unescape(q).strip()
        opts = re.findall(r'<li[^>]*id="([a-d])-\d+"[^>]*>(.*?)</li>', m.group(2), re.S | re.I)
        d = {}
        for letter, o in opts:
            ot = html.unescape(re.sub(r'<[^>]+>', '', o)).strip()
            if ot:
                d[letter.lower()] = ot
        if len(d) < 2 or not q:
            continue
        corr = m.group(3).lower()
        if corr not in d:
            continue
        letters = ['a', 'b', 'c', 'd']
        cands = [d.get(l, '') for l in letters]
        if not all(cands):
            continue
        rows.append({
            'question': q,
            'options': [x if x else '' for x in cands],
            'correct': letters.index(corr) + 1,
        })
    return rows


def has_next(h):
    if re.search(r"href=[\"'][^\"']*page=\d+[\"'][^>]*>\s*Next\s*<", h, re.I):
        return True
    nums = [int(x) for x in re.findall(r'>(\d+)</a>', h)]
    return bool(nums and max(nums) > 1)


def fetch_chapter(ch_url):
    rows = []
    error = None
    try:
        r = requests.get(ch_url, headers=H, timeout=40)
        if r.status_code != 200:
            return [], 'http' + str(r.status_code)
        rows += parse_page(r.text)
        pg = 2
        while has_next(r.text) and pg <= MAXPAGE:
            r = requests.get(ch_url + ('&' if '?' in ch_url else '?') + f'page={pg}', headers=H, timeout=40)
            if r.status_code != 200:
                break
            got = parse_page(r.text)
            if not got:
                break
            rows += got
            pg += 1
            time.sleep(0.15)
    except Exception as e:
        error = str(e)[:120]
    return rows, error


def main():
    home_html = requests.get(HOME, headers=H, timeout=40).text
    hubs = [u for u in collect_links(home_html) if is_hub(u)]
    chapters = set()
    for hb in hubs[:MAXHUBS]:
        try:
            h = requests.get(hb, headers=H, timeout=40).text
            for u in collect_links(h):
                if not is_hub(u) and '?' not in u and '.php' in u:
                    chapters.add(u)
        except Exception:
            continue
        time.sleep(0.05)
    print('hubs:', len(hubs), 'chapter urls:', len(chapters), flush=True)
    done = set()
    if os.path.exists(PROG):
        for ln in open(PROG, encoding='utf-8'):
            done.add(ln.strip())
    todo = [c for c in sorted(chapters) if c not in done]
    print('todo:', len(todo), flush=True)
    if not todo:
        return
    lock = threading.Lock()
    fw = open(DATA, 'a', encoding='utf-8')
    pg = open(PROG, 'a', encoding='utf-8')
    stats = {'rows': 0, 'ok': 0, 'err': 0}
    t0 = time.time()
    with ThreadPoolExecutor(max_workers=NTHREADS) as ex:
        futs = {ex.submit(fetch_chapter, c): c for c in todo}
        for i, f in enumerate(as_completed(futs), 1):
            ch = futs[f]
            rows, err = f.result()
            with lock:
                subj = subj_of(ch)
                topic = ch.rsplit('/', 1)[-1].split('.')[0].replace('-', ' ').title()
                for rw in rows:
                    fw.write(json.dumps({
                        'subject': subj, 'question': rw['question'], 'options': rw['options'],
                        'correct': rw['correct'], 'topic': topic, 'source': 'mcqlearn',
                        'url': ch, 'explanation': '',
                    }, ensure_ascii=False) + '\n')
                if err:
                    stats['err'] += 1
                else:
                    stats['ok'] += 1
                stats['rows'] += len(rows)
                pg.write(ch + '\n')
                if i % 20 == 0:
                    print(f'[{i}/{len(todo)}] rows={stats["rows"]} ok={stats["ok"]} err={stats["err"]} elapsed={time.time()-t0:.0f}s', flush=True)
    fw.close()
    pg.close()
    print('DONE', stats, flush=True)


if __name__ == '__main__':
    main()