File "fetch_pakmcqs.py"

Full path: /home/algopkco/public_html/scraper/fetch_pakmcqs.py
File size: 4.46 B (4.46 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8

Download   Open   Edit   Advanced Editor &nnbsp; Back

import json
import re
import html
import time
import random
import threading
import os

import requests

H = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/126.0 Safari/537.36'}
BASE = 'https://pakmcqs.com/category/{}/'
DATA = r'D:\xampp\htdocs\algo\scraper\data\pakmcqs_mcqs.jsonl'
PROG = r'D:\xampp\htdocs\algo\scraper\pakmcqs_progress.txt'
CATS = r'D:\xampp\htdocs\algo\scraper\data\pakmcqs_categories.json'
NTHREADS = 8

cats = json.load(open(CATS, encoding='utf-8'))
cats_by_id = {str(c['term_id']): c for c in cats}
parent = {str(c['term_id']): str(c['parent']) for c in cats}

def path_of(c):
    parts = [c['name']]
    p = str(c['parent'])
    depth = 0
    while p != '0' and p in cats_by_id and depth < 3:
        parts.append(cats_by_id[p]['name'])
        p = str(cats_by_id[p]['parent'])
        depth += 1
    return ' > '.join(reversed(parts))

def subject_of(c):
    p = path_of(c)
    parts = p.split(' > ')
    return parts[-1] if len(parts) == 1 else parts[-2]

def parse_block(b):
    m = re.search(r'<h2[^>]*>\s*<a[^>]*href="([^"]+)"[^>]*>(.*?)</a>', b, re.S)
    if not m:
        return None
    url = html.unescape(m.group(1))
    q = re.sub(r'<[^>]+>', '', m.group(2))
    q = html.unescape(q).strip()
    em = re.search(r'<div class="excerpt">(.*?)</div>', b, re.S)
    if not em:
        return None
    opt_html = em.group(1)
    qmark = re.search(r'<p>(.*?)</p>', opt_html, re.S)
    opt_html = qmark.group(1) if qmark else opt_html
    lines = re.split(r'<br\s*/?>', opt_html)
    opts = []
    corr = None
    for ln in lines:
        if not ln.strip():
            continue
        txt = re.sub(r'<[^>]+>', '', ln)
        txt = html.unescape(txt).strip()
        mm = re.match(r'^\(?([A-Da-d])\.?\s*(.*)$', txt)
        if not mm:
            continue
        letter = mm.group(1).upper()
        body = mm.group(2).strip()
        if re.search(r'<strong>', ln) and corr is None:
            corr = letter
        opts.append((letter, body))
    if len(opts) < 2 or corr is None:
        return None
    letters = ['A', 'B', 'C', 'D']
    d = {l: '' for l in letters}
    for l, b2 in opts:
        d[l] = b2
    return {'question': q, 'options': [d['A'], d['B'], d['C'], d['D']],
            'correct': letters.index(corr) + 1, 'url': url}

done = set()
if os.path.exists(PROG):
    for ln in open(PROG, encoding='utf-8'):
        done.add(ln.strip())
seen_urls = set()
if os.path.exists(DATA):
    for ln in open(DATA, encoding='utf-8'):
        try:
            seen_urls.add(json.loads(ln)['url'])
        except Exception:
            pass

lock = threading.Lock()
progress_log = open(PROG, 'a', encoding='utf-8')
fj = open(DATA, 'a', encoding='utf-8')

def crawl(c):
    slug = c['slug']
    if slug in done:
        return
    pages = max(1, (int(c.get('count', 0)) + 9) // 10)
    if pages > 300:
        pages = 300
    new_rows = 0
    for pg in range(1, pages + 1):
        url = BASE.format(slug) if pg == 1 else BASE.format(slug) + f'page/{pg}/'
        try:
            r = requests.get(url, headers=H, timeout=45)
            if r.status_code != 200:
                break
            t = r.text
            blocks = re.split(r'<article class="l-post', t)[1:]
            if not blocks:
                break
            for b in blocks:
                row = parse_block(b)
                if not row:
                    continue
                with lock:
                    if row['url'] in seen_urls:
                        continue
                    seen_urls.add(row['url'])
                    row['subject'] = subject_of(c)
                    row['topic'] = path_of(c)
                    row['source'] = 'pakmcqs'
                    row['explanation'] = ''
                    fj.write(json.dumps(row, ensure_ascii=False) + '\n')
                    fj.flush()
                    new_rows += 1
        except Exception as e:
            time.sleep(3)
        time.sleep(random.uniform(0.25, 0.8))
    with lock:
        progress_log.write(slug + '\n')
        progress_log.flush()
    print(f'{slug}: pages={pages} new={new_rows}', flush=True)

todos = [c for c in cats if int(c.get('count', 0)) > 0]
random.shuffle(todos)
print(f'cats to crawl: {len(todos)}', flush=True)

def worker():
    while True:
        with lock:
            if not todos:
                return
            c = todos.pop()
        crawl(c)

threads = [threading.Thread(target=worker) for _ in range(NTHREADS)]
for th in threads:
    th.start()
for th in threads:
    th.join()
print('ALL DONE', flush=True)