File "fetch_learncbse.py"
Full path: /home/algopkco/public_html/scraper/fetch_learncbse.py
File
size: 8.74 B (8.74 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8
Download Open Edit Advanced Editor &nnbsp; Back
import json
import re
import html
import os
import time
import threading
from concurrent.futures import ThreadPoolExecutor, as_completed
import requests
from pypdf import PdfReader
H = {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36',
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
}
URLS = r'D:\xampp\htdocs\algo\scraper\data\learncbse_urls.txt'
DATA = r'D:\xampp\htdocs\algo\scraper\data\learncbse_mcqs.jsonl'
PROG = r'D:\xampp\htdocs\algo\scraper\learncbse_progress.txt'
PDF_DIR = r'D:\xampp\htdocs\algo\scraper\data\learncbse_pdf'
NTHREADS = 12
CHEM = re.compile(r'(?i)(chemical|reactions? and equations|acids?|bases?|salts?|carbon|organic|metals?|non-metals?|periodic|molecules?|atoms?|matters? in our surroundings|is matter around|solutions?|compounds?|combustion|petroleum|coal|bonding|hydrocarbon|chemistry|separation of substances|sorting materials|water a precious|air around|synthetic fibres|plastics|materials)+')
PHYS = re.compile(r'(?i)(light|reflection|refraction|optics|lenses?|mirrors?|electricity|electric current|circuits?|magnet|motion|force|work and energy|power|sound|heat|temperature|gravitation|pressure|waves?|human eye|current|resistance|voltage|friction|momentum|inertia|density|buoyancy|thermometer|expansion|conduction|convection|radiation|kinetic|potential energy|physics|universe|stars?|solar|floatation|weight|mass|measuring|measurement)+')
BIO = re.compile(r'(?i)(cells?|tissues?|organs?|reproduction|heredity|evolution|nutrition|respiration|photosynthesis|transportation|excretion|control and coordination|hormones?|diseases?|health|immunity|microorganisms?|plants?|animals?|human|food|digestion|blood|heart|ecosystem|biosphere|biodiversity|classification|taxonomy|genetics|dna|biology|life processes|life cycle|vegetation|organism|pollination|seed)+')
SOC = re.compile(r'(?i)(social science|sst|history|geograph|political science|civics|economics?|constitution|janapada|mahajanapada|dynasty|empire|revolt|independence|movement|nationalism|colonial|democracy|governance|maratha|vijayanagara|medieval|ancient|modern india|gupta|maurya|delhi sultanate|mughal|rural|urban|village|panchayat|district|election|parliament|judiciary)+')
CS = re.compile(r'(?i)(computer|informatics|internet|programming|algorithm|python|coding|data science|artificial intelligence|ai |robotics)+')
ENG = re.compile(r'(?i)(english|grammar|vocabulary|tense|parts of speech|preposition|conjunction)+')
MATH = re.compile(r'(?i)(maths?|mathematics|algebra|geometry|trigonometry|calculus|statistics|probability|linear equations|quadratic|polynomials?|number system|integers|fractions?|decimals?|ratio|proportion|percentage|profit|interest|mensuration|triangles?|circles?|coordinate|real numbers|exponents|powers|squares?|cubes?|sets?|functions?|graph|angles?|parallelogram|volume|surface area|pythagoras)+')
def subj_of(title):
if CS.search(title):
return 'Computer Science General Test'
if MATH.search(title) and 'science' not in title.lower():
return 'Mathematics'
if SOC.search(title):
return 'Social-Studies-Mcqs'
if ENG.search(title):
return 'English Mcqs'
if CHEM.search(title):
return 'Chemistry Mcqs'
if PHYS.search(title):
return 'Physics Mcqs'
if BIO.search(title):
return 'Biology Mcqs'
if re.search(r'(?i)science', title):
return 'School Science'
return 'General Knowledge'
def parse_mcq_text(text):
r = []
blocks = re.split(r'Question\s+\d+\.\s*', text)
for b in blocks:
b = re.sub(r'\bAnswer\s+Answer\b', 'Answer', b)
ans_letters = re.findall(r'Answer\s*:\s*\(([a-d])\)', b, re.I)
if not ans_letters:
continue
clean = re.sub(r'Answer\s*:\s*\([a-d]\)[^.()]*\.?', '', b, flags=re.I)
opts = re.findall(r'\(([a-d])\)\s*([^(].*?)(?=\s*\([a-d]\)\s|$)', clean, re.S)
if len(opts) < 2:
continue
q = re.sub(r'\s+', ' ', clean.split(' (a)')[0]).strip()
q = q.lstrip('> ').strip()
for cut in ['Online Test with Answers', 'MCQ Questions with Answers', ' with Answers', 'Answers ']:
i = q.find(cut)
if i != -1:
q = q[i + len(cut):]
break
q = q.strip().lstrip('. :,').strip()
if not q:
continue
letters = ['A', 'B', 'C', 'D']
d = {l: '' for l in letters}
for l, o in opts[:4]:
d[l.upper()] = (l.upper() + ') ' + re.sub(r'\s+', ' ', o).strip()).strip()
if not all(d.values()):
continue
corr = letters.index(ans_letters[0].upper()) + 1
r.append({'question': q, 'options': [d['A'], d['B'], d['C'], d['D']], 'correct': corr})
return r
def pdf_text(path):
try:
r = PdfReader(path)
return '\n'.join((p.extract_text() or '') for p in r.pages)
except Exception:
return ''
def fetch_post(url, test_only=False):
try:
rp = requests.get(url, headers=H, timeout=40)
if rp.status_code != 200:
return url, 'http' + str(rp.status_code), []
h = rp.text
m = re.search(r'<title>(.*?)</title>', h, re.S)
title = html.unescape(re.sub(r'<[^>]+>', '', m.group(1))).strip() if m else url
title = re.sub(r'\s*\|\s*LearnCBSE.*$', '', title).strip()
m = re.search(r'class="entry-content"(.*?)(?=<footer|class="entry-footer")', h, re.S)
body = m.group(1) if m else h
body2 = re.sub(r'<script.*?</script>|<style.*?</style>', '', body, flags=re.S)
body3 = html.unescape(body2)
txt = re.sub(r'<[^>]+>', ' ', body3)
txt = html.unescape(re.sub(r'\s+', ' ', txt)).strip()
rows = parse_mcq_text(txt)
pds = re.findall(r'gview\s+file=(["\u201c\u201d]?)(https?://[^"\u201d\s>]+?\.pdf)\1', body3, re.I)
pds = [p[1] for p in pds] or re.findall(r'https?://[^"\u201d\s<>]+\.pdf', body3)
if test_only:
return url, title, len(rows), pds
if not rows and pds:
pd = pds[0]
fn = os.path.join(PDF_DIR, re.split(r'[/?]', pd)[-1])
try:
pr = requests.get(pd, headers=H, timeout=60)
if pr.status_code == 200 and len(pr.content) > 5000:
open(fn, 'wb').write(pr.content)
rows = parse_mcq_text(pdf_text(fn))
if not rows:
return url, 'pdf-no-text ' + title, []
except Exception as e:
return url, 'pdf-err ' + str(e), []
subj = subj_of(title)
out = []
seen = set()
for rw in rows:
qn = rw['question']
if qn in seen:
continue
seen.add(qn)
out.append({'subject': subj, 'question': qn, 'options': rw['options'],
'correct': rw['correct'], 'topic': title, 'source': 'learncbse',
'url': url, 'explanation': ''})
return url, title, out
except Exception as e:
return url, 'err ' + str(e), []
def main():
os.makedirs(PDF_DIR, exist_ok=True)
urls = [u.strip() for u in open(URLS, encoding='utf-8') if u.strip()]
done = set()
if os.path.exists(PROG):
for ln in open(PROG, encoding='utf-8'):
done.add(ln.strip())
done.update(u for u in urls if os.path.exists(os.path.join(PDF_DIR, 'x_' + re.split(r'[/?]', u)[-1])))
todo = [u for u in urls if u not in done]
print(f'total={len(urls)} done={len(done)} todo={len(todo)}', flush=True)
if not todo:
return
lock = threading.Lock()
fw = open(DATA, 'a', encoding='utf-8')
pg = open(PROG, 'a', encoding='utf-8')
stats = {'ok': 0, 'pdf': 0, 'pdf-notext': 0, 'err': 0, 'rows': 0}
t0 = time.time()
with ThreadPoolExecutor(max_workers=NTHREADS) as ex:
futs = {ex.submit(fetch_post, u): u for u in todo}
for i, f in enumerate(as_completed(futs), 1):
u, info, rows = f.result()
with lock:
if isinstance(rows, list) and rows:
for rw in rows:
fw.write(json.dumps(rw, ensure_ascii=False) + '\n')
stats['ok'] += 1
stats['rows'] += len(rows)
else:
if isinstance(info, str) and info.startswith('pdf-no-text'):
stats['pdf-notext'] += 1
else:
stats['err'] += 1
pg.write(u + '\n')
if i % 25 == 0:
el = time.time() - t0
print(f'[{i}/{len(todo)}] rows={stats["rows"]} ok={stats["ok"]} notext={stats["pdf-notext"]} err={stats["err"]} elapsed={el:.0f}s', flush=True)
fw.close()
pg.close()
print('DONE', stats)
if __name__ == '__main__':
main()