File "fetch_learncbse.py"

Full path: /home/algopkco/public_html/scraper/fetch_learncbse.py
File size: 8.74 B (8.74 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8

Download   Open   Edit   Advanced Editor &nnbsp; Back

import json
import re
import html
import os
import time
import threading
from concurrent.futures import ThreadPoolExecutor, as_completed

import requests
from pypdf import PdfReader

H = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
}
URLS = r'D:\xampp\htdocs\algo\scraper\data\learncbse_urls.txt'
DATA = r'D:\xampp\htdocs\algo\scraper\data\learncbse_mcqs.jsonl'
PROG = r'D:\xampp\htdocs\algo\scraper\learncbse_progress.txt'
PDF_DIR = r'D:\xampp\htdocs\algo\scraper\data\learncbse_pdf'
NTHREADS = 12

CHEM = re.compile(r'(?i)(chemical|reactions? and equations|acids?|bases?|salts?|carbon|organic|metals?|non-metals?|periodic|molecules?|atoms?|matters? in our surroundings|is matter around|solutions?|compounds?|combustion|petroleum|coal|bonding|hydrocarbon|chemistry|separation of substances|sorting materials|water a precious|air around|synthetic fibres|plastics|materials)+')
PHYS = re.compile(r'(?i)(light|reflection|refraction|optics|lenses?|mirrors?|electricity|electric current|circuits?|magnet|motion|force|work and energy|power|sound|heat|temperature|gravitation|pressure|waves?|human eye|current|resistance|voltage|friction|momentum|inertia|density|buoyancy|thermometer|expansion|conduction|convection|radiation|kinetic|potential energy|physics|universe|stars?|solar|floatation|weight|mass|measuring|measurement)+')
BIO = re.compile(r'(?i)(cells?|tissues?|organs?|reproduction|heredity|evolution|nutrition|respiration|photosynthesis|transportation|excretion|control and coordination|hormones?|diseases?|health|immunity|microorganisms?|plants?|animals?|human|food|digestion|blood|heart|ecosystem|biosphere|biodiversity|classification|taxonomy|genetics|dna|biology|life processes|life cycle|vegetation|organism|pollination|seed)+')
SOC = re.compile(r'(?i)(social science|sst|history|geograph|political science|civics|economics?|constitution|janapada|mahajanapada|dynasty|empire|revolt|independence|movement|nationalism|colonial|democracy|governance|maratha|vijayanagara|medieval|ancient|modern india|gupta|maurya|delhi sultanate|mughal|rural|urban|village|panchayat|district|election|parliament|judiciary)+')
CS = re.compile(r'(?i)(computer|informatics|internet|programming|algorithm|python|coding|data science|artificial intelligence|ai |robotics)+')
ENG = re.compile(r'(?i)(english|grammar|vocabulary|tense|parts of speech|preposition|conjunction)+')
MATH = re.compile(r'(?i)(maths?|mathematics|algebra|geometry|trigonometry|calculus|statistics|probability|linear equations|quadratic|polynomials?|number system|integers|fractions?|decimals?|ratio|proportion|percentage|profit|interest|mensuration|triangles?|circles?|coordinate|real numbers|exponents|powers|squares?|cubes?|sets?|functions?|graph|angles?|parallelogram|volume|surface area|pythagoras)+')


def subj_of(title):
    if CS.search(title):
        return 'Computer Science General Test'
    if MATH.search(title) and 'science' not in title.lower():
        return 'Mathematics'
    if SOC.search(title):
        return 'Social-Studies-Mcqs'
    if ENG.search(title):
        return 'English Mcqs'
    if CHEM.search(title):
        return 'Chemistry Mcqs'
    if PHYS.search(title):
        return 'Physics Mcqs'
    if BIO.search(title):
        return 'Biology Mcqs'
    if re.search(r'(?i)science', title):
        return 'School Science'
    return 'General Knowledge'


def parse_mcq_text(text):
    r = []
    blocks = re.split(r'Question\s+\d+\.\s*', text)
    for b in blocks:
        b = re.sub(r'\bAnswer\s+Answer\b', 'Answer', b)
        ans_letters = re.findall(r'Answer\s*:\s*\(([a-d])\)', b, re.I)
        if not ans_letters:
            continue
        clean = re.sub(r'Answer\s*:\s*\([a-d]\)[^.()]*\.?', '', b, flags=re.I)
        opts = re.findall(r'\(([a-d])\)\s*([^(].*?)(?=\s*\([a-d]\)\s|$)', clean, re.S)
        if len(opts) < 2:
            continue
        q = re.sub(r'\s+', ' ', clean.split(' (a)')[0]).strip()
        q = q.lstrip('> ').strip()
        for cut in ['Online Test with Answers', 'MCQ Questions with Answers', ' with Answers', 'Answers ']:
            i = q.find(cut)
            if i != -1:
                q = q[i + len(cut):]
                break
        q = q.strip().lstrip('. :,').strip()
        if not q:
            continue
        letters = ['A', 'B', 'C', 'D']
        d = {l: '' for l in letters}
        for l, o in opts[:4]:
            d[l.upper()] = (l.upper() + ') ' + re.sub(r'\s+', ' ', o).strip()).strip()
        if not all(d.values()):
            continue
        corr = letters.index(ans_letters[0].upper()) + 1
        r.append({'question': q, 'options': [d['A'], d['B'], d['C'], d['D']], 'correct': corr})
    return r


def pdf_text(path):
    try:
        r = PdfReader(path)
        return '\n'.join((p.extract_text() or '') for p in r.pages)
    except Exception:
        return ''


def fetch_post(url, test_only=False):
    try:
        rp = requests.get(url, headers=H, timeout=40)
        if rp.status_code != 200:
            return url, 'http' + str(rp.status_code), []
        h = rp.text
        m = re.search(r'<title>(.*?)</title>', h, re.S)
        title = html.unescape(re.sub(r'<[^>]+>', '', m.group(1))).strip() if m else url
        title = re.sub(r'\s*\|\s*LearnCBSE.*$', '', title).strip()
        m = re.search(r'class="entry-content"(.*?)(?=<footer|class="entry-footer")', h, re.S)
        body = m.group(1) if m else h
        body2 = re.sub(r'<script.*?</script>|<style.*?</style>', '', body, flags=re.S)
        body3 = html.unescape(body2)
        txt = re.sub(r'<[^>]+>', ' ', body3)
        txt = html.unescape(re.sub(r'\s+', ' ', txt)).strip()
        rows = parse_mcq_text(txt)
        pds = re.findall(r'gview\s+file=(["\u201c\u201d]?)(https?://[^"\u201d\s>]+?\.pdf)\1', body3, re.I)
        pds = [p[1] for p in pds] or re.findall(r'https?://[^"\u201d\s<>]+\.pdf', body3)
        if test_only:
            return url, title, len(rows), pds
        if not rows and pds:
            pd = pds[0]
            fn = os.path.join(PDF_DIR, re.split(r'[/?]', pd)[-1])
            try:
                pr = requests.get(pd, headers=H, timeout=60)
                if pr.status_code == 200 and len(pr.content) > 5000:
                    open(fn, 'wb').write(pr.content)
                    rows = parse_mcq_text(pdf_text(fn))
                    if not rows:
                        return url, 'pdf-no-text ' + title, []
            except Exception as e:
                return url, 'pdf-err ' + str(e), []
        subj = subj_of(title)
        out = []
        seen = set()
        for rw in rows:
            qn = rw['question']
            if qn in seen:
                continue
            seen.add(qn)
            out.append({'subject': subj, 'question': qn, 'options': rw['options'],
                        'correct': rw['correct'], 'topic': title, 'source': 'learncbse',
                        'url': url, 'explanation': ''})
        return url, title, out
    except Exception as e:
        return url, 'err ' + str(e), []


def main():
    os.makedirs(PDF_DIR, exist_ok=True)
    urls = [u.strip() for u in open(URLS, encoding='utf-8') if u.strip()]
    done = set()
    if os.path.exists(PROG):
        for ln in open(PROG, encoding='utf-8'):
            done.add(ln.strip())
    done.update(u for u in urls if os.path.exists(os.path.join(PDF_DIR, 'x_' + re.split(r'[/?]', u)[-1])))
    todo = [u for u in urls if u not in done]
    print(f'total={len(urls)} done={len(done)} todo={len(todo)}', flush=True)
    if not todo:
        return
    lock = threading.Lock()
    fw = open(DATA, 'a', encoding='utf-8')
    pg = open(PROG, 'a', encoding='utf-8')
    stats = {'ok': 0, 'pdf': 0, 'pdf-notext': 0, 'err': 0, 'rows': 0}
    t0 = time.time()
    with ThreadPoolExecutor(max_workers=NTHREADS) as ex:
        futs = {ex.submit(fetch_post, u): u for u in todo}
        for i, f in enumerate(as_completed(futs), 1):
            u, info, rows = f.result()
            with lock:
                if isinstance(rows, list) and rows:
                    for rw in rows:
                        fw.write(json.dumps(rw, ensure_ascii=False) + '\n')
                    stats['ok'] += 1
                    stats['rows'] += len(rows)
                else:
                    if isinstance(info, str) and info.startswith('pdf-no-text'):
                        stats['pdf-notext'] += 1
                    else:
                        stats['err'] += 1
                pg.write(u + '\n')
                if i % 25 == 0:
                    el = time.time() - t0
                    print(f'[{i}/{len(todo)}] rows={stats["rows"]} ok={stats["ok"]} notext={stats["pdf-notext"]} err={stats["err"]} elapsed={el:.0f}s', flush=True)
    fw.close()
    pg.close()
    print('DONE', stats)


if __name__ == '__main__':
    main()