File "fetch_tpointtech.py"
Full path: /home/algopkco/public_html/scraper/fetch_tpointtech.py
File
size: 11.3 B (11.3 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8
Download Open Edit Advanced Editor &nnbsp; Back
import html as H
import json
import os
import re
import time
import requests
BASE = os.path.dirname(os.path.abspath(__file__))
DATA = os.path.join(BASE, "data")
MENU_FILE = os.path.join(DATA, "tpointtech_menu.txt")
OUT_FILE = os.path.join(DATA, "tpointtech_mcqs.jsonl")
PROGRESS_FILE = os.path.join(DATA, "tpointtech_progress.txt")
HEADERS = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36"
}
SUBJECTS = {
"cpp-mcq": "C++ Programming",
"c-language-mcq": "C Programming",
"python-mcq": "Python Programming",
"java-mcq": "Java Programming",
"java-multithreading-mcqs": "Java Programming",
"thread-priority-mcqs-in-java": "Java Programming",
"javascript-mcq": "JavaScript",
"jquery-mcq": "jQuery",
"dbms-mcq": "Database Management System",
"mcqs-of-entity-relationship-diagram": "Database Management System",
"mcqs-on-normalization": "Database Management System",
"mcqs-on-relational-algebra": "Database Management System",
"mcqs-on-relational-calculus": "Database Management System",
"mcqs-on-transactions-and-concurrency-control": "Database Management System",
"sql-mcq": "SQL",
"html-mcq": "HTML",
"css-mcq": "CSS",
"bootstrap-mcq": "CSS",
"bootstrap-4-mcq": "CSS",
"data-structure-mcq": "Data Structures And Algorithms",
"mcqs-on-sorting-techniques": "Data Structures And Algorithms",
"mcqs-on-splay-tree": "Data Structures And Algorithms",
"mcqs-on-hamiltonian-graph": "Data Structures And Algorithms",
"mcqs-on-knapsack-algorithm": "Data Structures And Algorithms",
"mcq-on-divide-and-conquer-algorithm": "Data Structures And Algorithms",
"mcqs-on-greedy-algorithm": "Data Structures And Algorithms",
"mcqs-on-n-queens-problem": "Data Structures And Algorithms",
"mcqs-on-topological-sorting": "Data Structures And Algorithms",
"operating-system-mcq": "Operating System",
"computer-network-mcq": "Computer Networks",
"oops-mcq": "Object Oriented Programming",
"digital-electronics-mcq": "Digital Electronics",
"compiler-design-mcq": "Compiler Design",
"software-engineering-mcq": "Software Engineering",
"software-testing-mcq": "Software Engineering",
"computer-fundamental-mcq": "Computer Fundamentals",
"computer-awareness-mcq-for-bank-exam": "Computer Fundamentals",
"mcq-on-computer-hardware-for-bank-exam": "Computer Fundamentals",
"mcq-on-computer-software-for-bank-exams": "Computer Fundamentals",
"artificial-intelligence-mcq": "Artificial Intelligence",
"data-mining-mcq": "Data Mining",
"cloud-computing-mcq": "Cloud Computing",
"iot-mcq": "IoT",
"cyber-security-mcq": "Cyber Security",
"mobile-computing-mcq": "Mobile Computing",
"computer-architecture-mcq": "Computer Architecture",
"computer-graphics-mcq": "Computer Graphics",
"digital-image-processing-mcq": "Digital Image Processing",
"digital-communication-mcq": "Electronics Engineering",
"digital-signal-processing-mcq": "Electronics Engineering",
"soft-computing-mcq": "Soft Computing",
"embedded-systems-mcq": "Embedded Systems",
"electrical-mcq": "Electrical Engineering",
"transformer-mcq": "Electrical Engineering",
"control-system-mcq": "Electrical Engineering",
"power-electronics-mcq": "Electrical Engineering",
"mechanical-engineering-mcq": "Mechanical Engineering",
"fluid-mechanics-mcq": "Mechanical Engineering",
"engineering-mechanics-mcq": "Mechanical Engineering",
"civil-engineering-mcq": "Civil Engineering",
"genetics-algorithm-mcq": "Genetic Algorithms",
"discrete-mathematics-mcq": "Mathematics",
"probability-mcq": "Mathematics",
"statistics-mcq": "Mathematics",
"number-system-mcq": "Mathematics",
"trigonometry-mcq": "Mathematics",
"reasoning-mcq": "Reasoning",
"gk-mcq": "General Knowledge",
"general-science-mcq": "General Knowledge",
"ancient-history-mcq": "General Knowledge",
"indian-constitution-mcq": "General Knowledge",
"human-rights-mcq": "General Knowledge",
"mcq-of-important-discoveries-and-invention": "General Knowledge",
"mcq-on-vitamins-and-nutrition": "General Knowledge",
"environmental-science-mcq": "Environmental Science",
"environmental-studies-mcq": "Environmental Science",
"life-processes-mcq": "School Science",
"class-10th-science-mcq": "School Science",
"class-9th-science-mcq": "School Science",
"class-12-physics-mcq": "School Science",
"biotechnology-mcq": "Biotechnology",
"psychology-mcq": "Psychology",
"human-resource-management-mcq": "Management",
"research-methodology-mcq": "Management",
"english-grammar-mcq-for-competitive-exam": "English Grammar",
"powerpoint-mcq": "Microsoft Office",
"digital-marketing-mcq": "Digital Marketing",
}
SKIP_PAGES = {"mcqs-preparation"}
LETTERS = "ABCDEFGHIJKLMNOPQRSTUVWXYZ"
session = requests.Session()
session.headers.update(HEADERS)
def get(url, tries=5):
for i in range(tries):
try:
r = session.get(url, timeout=30)
if r.status_code == 200:
return r.text
print(" HTTP", r.status_code, url, flush=True)
if r.status_code == 404:
return None
except Exception as e:
print(" err", e, "retry", i, flush=True)
time.sleep(1 + i * 2)
return None
def clean(s):
s = re.sub(r"<br\s*/?>", " ", s)
s = re.sub(r"<[^>]+>", "", s)
s = H.unescape(s)
s = re.sub(r"\xa0", " ", s)
s = re.sub(r"[ \t]+", " ", s)
s = re.sub(r"\n\s*\n+", "\n", s)
return s.strip()
PQ = re.compile(r'<p\s+class=["\']?pq["\']?>\s*', re.I)
def parse_page(html, page_url, subject, topic):
chunks = PQ.split(html)
out = []
for chunk in chunks[1:]:
ocut = chunk.find("<ol class=")
if ocut == -1:
ocut = chunk.find('<ol class="')
qpart = chunk if ocut == -1 else chunk[:ocut]
om = re.search(r'<ol\s+class=["\']?pointsa["\']?>(.*?)</ol>', chunk, re.S)
if om:
options = [clean(o) for o in re.findall(r"<li>(.*?)</li>", om.group(1), re.S)]
options = [o for o in options if o]
qtext = re.sub(r"^\d+\)?\.?\s*", "", clean(re.sub(r"</p>", " ", qpart))).strip()
else:
parts = re.split(r"<p>\s*([a-zA-Z])\.\s*([^<]*)</p>", qpart)
options = []
for pi in range(1, len(parts), 3):
lbl, inline, blk = parts[pi], parts[pi + 1], parts[pi + 2]
txt = ""
if inline.strip():
txt = clean(inline)
else:
tas = re.findall(r"<textarea[^>]*>(.*?)</textarea>", blk, re.S)
if tas:
txt = " / ".join(H.unescape(re.sub(r"<[^>]+>", "", t)).strip() for t in tas if t.strip())
else:
txt = clean(blk)
if txt:
options.append(txt)
qtext = re.sub(r"^\d+\)?\.?\s*", "", clean(re.sub(r"</p>", " ", parts[0]))).strip()
if not qtext:
continue
cm = re.search(r"<textarea[^>]*>(.*?)</textarea>", qpart, re.S) or re.search(
r"<pre[^>]*>(.*?)</pre>", qpart, re.S
)
if cm:
code = H.unescape(re.sub(r"<[^>]+>", "", cm.group(1))).strip()
if code:
qtext = qtext + "\n\n```\n" + code + "\n```"
if len(options) < 2:
continue
am = re.search(r'<div\s+class=["\']?testanswer["\']?[^>]*>(.*?)</div>', chunk, re.S)
if not am:
continue
amt = am.group(1)
amatch = re.search(r"Answer:\s*(?:<[^>]+>\s*)*([A-Z])\b", amt, re.I)
if not amatch:
amatch = re.search(r"Explanation:\s*(?:<[^>]+>\s*)*([A-Z])\b", amt, re.I)
if not amatch:
continue
ci = LETTERS.upper().find(amatch.group(1).upper())
if ci < 0 or ci >= len(options):
continue
emi = amt.find("Explanation:")
expl = clean(amt[emi + len("Explanation:"):]) if emi != -1 else ""
out.append(
{
"subject": subject,
"topic": topic,
"question": qtext,
"options": options,
"correct": ci + 1,
"explanation": expl,
"source_url": page_url,
}
)
return out
def topic_name(title, page_url):
t = H.unescape(re.sub(r"<[^>]+>", "", title or "")).strip()
t = re.sub(r"MCQ'?s|MCQs|MCQ", "", t, flags=re.I)
t = re.sub(r"\s*Part\s*\d+$", "", t, flags=re.I)
t = re.sub(r"\s+", " ", t).strip()
return t or page_url
def main():
menu = []
if os.path.exists(MENU_FILE):
with open(MENU_FILE, encoding="utf-8") as f:
for ln in f:
parts = ln.rstrip("\n").split("\t")
if len(parts) == 2:
menu.append((parts[0], parts[1]))
else:
print("menu file missing, crawling cpp-mcq only")
menu = [("https://www.tpointtech.com/cpp-mcq", "C++ MCQ")]
done = set()
if os.path.exists(PROGRESS_FILE):
with open(PROGRESS_FILE, encoding="utf-8") as f:
done = {ln.strip() for ln in f}
fout = open(OUT_FILE, "a", encoding="utf-8")
total = 0
seen_questions = set()
if os.path.exists(OUT_FILE):
with open(OUT_FILE, encoding="utf-8") as f:
for ln in f:
try:
it = json.loads(ln)
seen_questions.add(it["question"])
except Exception:
pass
pages = []
for url, title in menu:
m = re.search(r"tpointtech\.com/([^/]+)", url)
if not m or m.group(1) in SKIP_PAGES:
continue
base = m.group(1)
subj = None
for k, v in SUBJECTS.items():
if base == k or base.startswith(k + "-part"):
subj = v
break
if not subj:
continue
pages.append((url, base, title, subj))
for i in range(2, 10):
variant_a = f"{base}-part-{i}"
variant_b = f"{base}-part{i}"
for v in (variant_a, variant_b):
if v == base:
continue
pages.append((f"https://www.tpointtech.com/{v}", v, f"{title} Part {i}", subj))
print("pages to crawl:", len(pages), flush=True)
new = 0
for url, slug, title, subj in pages:
if url in done:
continue
try:
html = get(url)
if not html:
continue
qs = parse_page(html, url, subj, topic_name(title, slug))
for it in qs:
if it["question"] in seen_questions:
continue
seen_questions.add(it["question"])
fout.write(json.dumps(it, ensure_ascii=False) + "\n")
new += 1
total += len(qs)
except Exception as e:
print("ERR on", slug, e, flush=True)
continue
with open(PROGRESS_FILE, "a", encoding="utf-8") as pf:
pf.write(url + "\n")
fout.flush()
print(f"{slug}: {len(qs)} q (new {new}, total {total})", flush=True)
time.sleep(0.6)
fout.close()
print("DONE pages done:", len(done), "questions total:", total, "new:", new)
if __name__ == "__main__":
main()