File "fetch_pakmcqs.py"
Full path: /home/algopkco/public_html/scraper/fetch_pakmcqs.py
File
size: 4.46 B (4.46 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8
Download Open Edit Advanced Editor &nnbsp; Back
import json
import re
import html
import time
import random
import threading
import os
import requests
H = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/126.0 Safari/537.36'}
BASE = 'https://pakmcqs.com/category/{}/'
DATA = r'D:\xampp\htdocs\algo\scraper\data\pakmcqs_mcqs.jsonl'
PROG = r'D:\xampp\htdocs\algo\scraper\pakmcqs_progress.txt'
CATS = r'D:\xampp\htdocs\algo\scraper\data\pakmcqs_categories.json'
NTHREADS = 8
cats = json.load(open(CATS, encoding='utf-8'))
cats_by_id = {str(c['term_id']): c for c in cats}
parent = {str(c['term_id']): str(c['parent']) for c in cats}
def path_of(c):
parts = [c['name']]
p = str(c['parent'])
depth = 0
while p != '0' and p in cats_by_id and depth < 3:
parts.append(cats_by_id[p]['name'])
p = str(cats_by_id[p]['parent'])
depth += 1
return ' > '.join(reversed(parts))
def subject_of(c):
p = path_of(c)
parts = p.split(' > ')
return parts[-1] if len(parts) == 1 else parts[-2]
def parse_block(b):
m = re.search(r'<h2[^>]*>\s*<a[^>]*href="([^"]+)"[^>]*>(.*?)</a>', b, re.S)
if not m:
return None
url = html.unescape(m.group(1))
q = re.sub(r'<[^>]+>', '', m.group(2))
q = html.unescape(q).strip()
em = re.search(r'<div class="excerpt">(.*?)</div>', b, re.S)
if not em:
return None
opt_html = em.group(1)
qmark = re.search(r'<p>(.*?)</p>', opt_html, re.S)
opt_html = qmark.group(1) if qmark else opt_html
lines = re.split(r'<br\s*/?>', opt_html)
opts = []
corr = None
for ln in lines:
if not ln.strip():
continue
txt = re.sub(r'<[^>]+>', '', ln)
txt = html.unescape(txt).strip()
mm = re.match(r'^\(?([A-Da-d])\.?\s*(.*)$', txt)
if not mm:
continue
letter = mm.group(1).upper()
body = mm.group(2).strip()
if re.search(r'<strong>', ln) and corr is None:
corr = letter
opts.append((letter, body))
if len(opts) < 2 or corr is None:
return None
letters = ['A', 'B', 'C', 'D']
d = {l: '' for l in letters}
for l, b2 in opts:
d[l] = b2
return {'question': q, 'options': [d['A'], d['B'], d['C'], d['D']],
'correct': letters.index(corr) + 1, 'url': url}
done = set()
if os.path.exists(PROG):
for ln in open(PROG, encoding='utf-8'):
done.add(ln.strip())
seen_urls = set()
if os.path.exists(DATA):
for ln in open(DATA, encoding='utf-8'):
try:
seen_urls.add(json.loads(ln)['url'])
except Exception:
pass
lock = threading.Lock()
progress_log = open(PROG, 'a', encoding='utf-8')
fj = open(DATA, 'a', encoding='utf-8')
def crawl(c):
slug = c['slug']
if slug in done:
return
pages = max(1, (int(c.get('count', 0)) + 9) // 10)
if pages > 300:
pages = 300
new_rows = 0
for pg in range(1, pages + 1):
url = BASE.format(slug) if pg == 1 else BASE.format(slug) + f'page/{pg}/'
try:
r = requests.get(url, headers=H, timeout=45)
if r.status_code != 200:
break
t = r.text
blocks = re.split(r'<article class="l-post', t)[1:]
if not blocks:
break
for b in blocks:
row = parse_block(b)
if not row:
continue
with lock:
if row['url'] in seen_urls:
continue
seen_urls.add(row['url'])
row['subject'] = subject_of(c)
row['topic'] = path_of(c)
row['source'] = 'pakmcqs'
row['explanation'] = ''
fj.write(json.dumps(row, ensure_ascii=False) + '\n')
fj.flush()
new_rows += 1
except Exception as e:
time.sleep(3)
time.sleep(random.uniform(0.25, 0.8))
with lock:
progress_log.write(slug + '\n')
progress_log.flush()
print(f'{slug}: pages={pages} new={new_rows}', flush=True)
todos = [c for c in cats if int(c.get('count', 0)) > 0]
random.shuffle(todos)
print(f'cats to crawl: {len(todos)}', flush=True)
def worker():
while True:
with lock:
if not todos:
return
c = todos.pop()
crawl(c)
threads = [threading.Thread(target=worker) for _ in range(NTHREADS)]
for th in threads:
th.start()
for th in threads:
th.join()
print('ALL DONE', flush=True)