import html as H import json import os import re import time import requests BASE = os.path.dirname(os.path.abspath(__file__)) DATA = os.path.join(BASE, "data") OUT = os.path.join(DATA, "t4_mcqs.jsonl") PROG = os.path.join(BASE, "t4_progress2.txt") LOG = os.path.join(BASE, "t4_run2.log") HEADERS = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36", } s = requests.Session() s.headers.update(HEADERS) SLUGS = [l.strip() for l in open(os.path.join(BASE, "t4cpp_progress.txt"), encoding="utf-8") if l.strip()] CPP_SLUGS = { "advanced-c-plus-plus-mcqs", "arrays-mcqs-questions-answers-c", "c-array-solved-mcqs-questions-answers", "c-mcqs", "c-standard-library-mcqs-questions-answers", "classes-and-inheritance-mcqs-in-c-oop", "friend-function-mcqs", "function-arguments-pass-by-value-reference-c-mcqs", "inline-function-mcqs", "mcqs-of-introduction-to-programming", "multi-dimensional-arrays-mcqs", "one-dimensional-and-multi-dimensional-arrays-c-mcqs", "one-dimensional-arrays-mcqs", "operator-overloading-solved-mcqs-oop", "parameter-passing-mechanisms-call-by-value-referencemcqs", "pointers-and-arrays-c-mcqs", "pointers-solved-mcqs-questions-answers", "pointers-to-functions-c-mcqs", "pointers-to-pointers-c-mcqs", "polymorphism-mcqs-in-object-oriented-programmingoop", "programming-c-mcqs-for-programming-competition", } CS_SUBJECTS = { "automata-theory-mcqs": "Computer Science General Test", "compiler-construction-mcqs": "Computer Science General Test", "computer-graphics-solved-mcqs-questions-answers": "Computer Science General Test", "cyber-crime-solved-mcqs-questions-answers": "Computer Science General Test", "deadlock-mcqs-questions-answers": "Operating Systems", "digital-image-processing-mcqs": "Computer Science General Test", "distributed-database-architecture-mcqs-mcqs-in-dbms": "Database Management Systems", "html-mcqs-test-solved-questions-answers": "Web Technologies", "html-solved-mcqs": "Web Technologies", "css-mcqs-for-web-developer-and-managers-jobs-test": "Web Technologies", "introduction-to-computing-itc-mcqs": "Computer Science General Test", "java-online-test-mcqs-questions-answers": "Java Programming", "management-information-system-mis-solved-mcqs-with-answers-pdf": "Management Information Systems", "mcqs-analysis-of-algorithms-for-jobs-test-solved": "Design And Analysis Of Algorithms", "mcqs-computer-science-core-courses": "Computer Science General Test", "mcqs-elective-courses-computer-science": "Computer Science General Test", "mcqs-on-viruses-and-computer-security": "Computer Security", "mobile-android-applications-mcqs": "Mobile Application Development", "mpi-mcqs-message-passing-interface-mcqs": "Computer Science General Test", "network-layer-osi-model-solved-mcqs": "Computer Networks", "networking-mcqs-storage-solutions-cloud-computing-mcqs-data-center-technologies-mcqs": "Computer Networks", "php-mcqs-solved-questions-answers-for-web-developers-and-managers": "Web Technologies", "social-networks-mcqs-solved-questions-answers": "Computer Science General Test", "system-programming-mcqs": "Computer Science General Test", "virtual-memory-mcqs-questions-answers-in-operating-systems": "Operating Systems", "technical-report-writing-solved-mcqs": "Computer Science General Test", "technology-management-mcqs": "Computer Science General Test", "computer-science-mcqs-homepage": "Computer Science General Test", "computer-science-mcqs-leaks-pdf-ebook-by-fazal-rehman-shamil": "Computer Science General Test", "software-engineering-mcqs": "Software Engineering", } CS_KEYWORDS = re.compile( r"(computer|software|hardware|programming|algorithm|data|network|database|dbms|html|css|php|java|javascript|" r"python|android|operating|os-|web|sql|compiler|automata|digital|security|graphic|system|linux|cloud|" r"information|multimedia|micro|processor|artificial|cryptography|e-?commerce|oop|mcq-of|ic3|ict|itc|pract|" r"mobile|server|stack|queue|linked|sorting|searching|binary|recursion|hci|cpu|memory|input-output|" r"windows|ms-office|office)", re.I, ) def norm(t): t = H.unescape(t) t = re.sub(r"<[^>]+>", "", t) t = re.sub(r"\s+", " ", t) return t.strip() def get(url, tries=4): for i in range(tries): try: r = s.get(url, timeout=30) if r.status_code == 200: return r.text except Exception: pass time.sleep(1 + i) return None def parse_format_a(html): rows = [] for b in re.findall(r'<div class="question-container">(.*?)</div>', html, re.S): qm = re.search(r"question-text\">(.*?)</strong>", b, re.S) if not qm: continue q = re.sub(r"^\d+\.\s*", "", norm(qm.group(1))) if len(q) < 10 or q.endswith("Back"): continue opts, ci = [], -1 for lb in re.findall(r"<label>(.*?)</label>", b, re.S): vm = re.search(r'value="(.*?)"', lb, re.S) om = re.search(r"onclick=\"checkAnswer\('[^']*', ?'(.*?)'\)\"", lb, re.S) if not vm: continue label = norm(lb) m = re.search(r"\(([A-E])\)\s*(.*)$", label, re.S) if m: opts.append(norm(m.group(2))) else: opts.append(norm(vm.group(1))) if om and ci < 0: ans = norm(om.group(1)) for i, o in enumerate(opts): if o == ans: ci = i break if len(opts) >= 2 and 0 <= ci < len(opts): rows.append((q, opts, ci)) return rows def parse_format_c(html): rows = [] for ch in re.split(r"Q#\s*\d+\s*:", html)[1:]: ans_part = None if "Answer:" in ch: body, ans_part = ch.split("Answer:", 1) else: body = ch pieces = re.split(r"\(([A-E])\)", body) q = norm(pieces[0]) q = re.split(r"Answer:", q)[0] if len(q) < 5: continue opts = [] for i in range(1, len(pieces) - 1, 2): opts.append(norm(pieces[i + 1])) opts = [o for o in opts if o] ci = -1 if ans_part: am = re.search(r"^\s*\(([A-E])\)", ans_part) if am: ci = "ABCDE".find(am.group(1)) if len(opts) >= 2 and 0 <= ci < len(opts): rows.append((q, opts, ci)) return rows def parse_format_d(html): clean = re.sub(r"<script.*?</script>|<style.*?</style>", "", html, flags=re.S) clean = re.sub(r"<[^>]+>", " ", clean) clean = H.unescape(clean) clean = re.sub(r"\s+", " ", clean) rows = [] spans = [] for m in re.finditer(r"Answer\s*:\s*([a-eA-E])\s*\)", clean): spans.append((m.start(), m.group(1))) for i, (pos, letter) in enumerate(spans): end = spans[i + 1][0] if i + 1 < len(spans) else min(pos + 400, len(clean)) seg = clean[max(pos - 900, 0):end] qm = re.search(r"([A-Za-z0-9][^A-Za-z]{0,30}\?)\s", seg) if not qm: continue q = norm(qm.group(1)) if len(q) < 10: continue tail = seg[qm.end():] opts = [norm(x) for x in re.split(r"[a-eA-E]\s*\)\s*", tail) if norm(x)] ci = "abcde".find(letter.lower()) if len(opts) >= 2 and 0 <= ci < len(opts): rows.append((q, opts, ci)) return rows def parse_questions(html): rows = parse_format_a(html) if not rows: rows = parse_format_c(html) if not rows: rows = parse_format_d(html) return rows def classify(slug): if slug in CPP_SLUGS: return "C++ Programming", "T4Tutorials C++ MCQs" if slug in CS_SUBJECTS: return CS_SUBJECTS[slug], "T4Tutorials " + slug.replace("-", " ").title() if CS_KEYWORDS.search(slug): return "Computer Science General Test", "T4Tutorials " + slug.replace("-", " ").title() return None, None def main(): done = set() if os.path.exists(PROG): done = set(l.strip() for l in open(PROG, encoding="utf-8") if l.strip()) seen = {} if os.path.exists(OUT): for ln in open(OUT, encoding="utf-8"): try: j = json.loads(ln) seen.setdefault(norm(j["question"]), j) except Exception: pass fout = open(OUT, "a", encoding="utf-8") flog = open(LOG, "a", encoding="utf-8") crash = open(os.path.join(BASE, "t4_crash.log"), "a", encoding="utf-8") new = 0 for slug in SLUGS: if slug in done: continue url = f"https://t4tutorials.com/{slug}/" try: html = get(url) if not html: flog.write(f"{slug}: NO HTML\n") flog.flush() with open(PROG, "a", encoding="utf-8") as f: f.write(slug + "\n") continue subj, topic = classify(slug) import threading res = {} thr = threading.Thread(target=lambda: res.update(rows=parse_questions(html))) thr.start() thr.join(60) if thr.is_alive(): flog.write(f"{slug}: PARSE TIMEOUT\n") flog.flush() rows = [] else: rows = res.get("rows", []) cnt = 0 for q, opts, ci in rows: k = norm(q) if k in seen: continue it = { "subject": subj or slug, "topic": topic or slug, "question": q, "options": opts, "correct": ci + 1, "explanation": "", "source_url": url, } seen[k] = it fout.write(json.dumps(it, ensure_ascii=False) + "\n") cnt += 1 new += 1 fout.flush() flog.write(f"{slug}: {len(rows)} parsed, {cnt} new\n") flog.flush() with open(PROG, "a", encoding="utf-8") as f: f.write(slug + "\n") print(f"{slug}: {len(rows)} parsed, {cnt} new (total new {new})", flush=True) except Exception as e: import traceback crash.write(f"CRASH {url}: {e}\n{traceback.format_exc()}\n") crash.flush() with open(PROG, "a", encoding="utf-8") as f: f.write(slug + "\n") print(f"{slug}: CRASH {e}", flush=True) time.sleep(0.4) fout.close() flog.close() print("DONE total new:", new) if __name__ == "__main__": main()