File "map_categories.py"
Full path: /home/algopkco/public_html/scraper/map_categories.py
File
size: 7.23 B (7.23 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8
Download Open Edit Advanced Editor &nnbsp; Back
import hashlib
import json
import os
import re
import sys
import time
BASE = os.path.dirname(os.path.abspath(__file__))
PARSED_FILE = os.path.join(BASE, "data", "parsed_mcqs.jsonl")
# category_id -> (subject_title, subject_code, subject_slug)
SUBJECT_MAP = {
1: ("World General Knowledge", "WGK102", "world-general-knowledge"),
37: ("World Current Affairs", "WCA001", "world-current-affairs"),
70: ("Pakistan Current Affairs", "PCA001", "pakistan-current-affairs"),
48: ("Pakistan Study General Test", "psgt11", "pakistan-study-general-test"),
39: ("Everyday Science General", "EDS001", "everyday-science-general"),
38: ("Islamic Studies General", "ISG001", "islamic-studies-general"),
50: ("Computer Science General Test", "CSGT909", "computer-science-general-test"),
44: ("General English Grammar", "ENGGT01", "general-english-grammar"),
45: ("Quantitative Aptitude Mathematics", "QAM102", "quantitative-aptitude-mathematics"),
41: ("Biology General Test", "BioGT", "biology-general-test"),
40: ("Chemistry General Test ", "ChemGT", "chemistry-general-test"),
46: ("Physics General Test", "phygt", "physics-general-test"),
740: ("Medical Sciences General", "MDG001", "medical-sciences-general"),
79: ("Economics General", "ECO001", "economics-general"),
856: ("Political Science General", "POL001", "political-science-general"),
521: ("English Literature General", "ELT001", "english-literature-general"),
742: ("Psychology General", "PSY001", "psychology-general"),
570: ("Sociology General", "SOC001", "sociology-general"),
109: ("Judiciary And Law", "JDL001", "judiciary-and-law"),
982: ("International Relations General", "IRG001", "international-relations-general"),
54: ("Marketing General", "MKT001", "marketing-general"),
53: ("HRM General", "HRM001", "hrm-general"),
52: ("Finance General", "FIN001", "finance-general"),
71: ("Accounting General", "ACC001", "accounting-general"),
80: ("Auditing General", "AUD001", "auditing-general"),
1025: ("Forestry General", "FOR001", "forestry-general"),
393: ("Agriculture General", "AGR001", "agriculture-general"),
76: ("Statistics General", "STA001", "statistics-general"),
147: ("Pedagogy General", "PED001", "pedagogy-general"),
166: ("Urdu General", "URD001", "urdu-general"),
320: ("Chemical Engineering General", "CHE001", "chemical-engineering-general"),
285: ("Mechanical Engineering General", "MEC001", "mechanical-engineering-general"),
240: ("Civil Engineering General", "CIV001", "civil-engineering-general"),
111: ("Electrical Engineering General", "ELE001", "electrical-engineering-general"),
304: ("Software Engineering General", "SWE001", "software-engineering-general"),
1066: ("Physical Education General", "PHE001", "physical-education-general"),
1064: ("Past Papers General", "PAP001", "past-papers-general"),
737: ("Judiciary And Law", "JDL001", "judiciary-and-law"),
743: ("Political Science General", "POL001", "political-science-general"),
855: ("Chemistry General Test ", "ChemGT", "chemistry-general-test"),
42: ("Computer Science General Test", "CSGT909", "computer-science-general-test"),
74: ("Management General", "MGT001", "management-general"),
75: ("Quantitative Aptitude Mathematics", "QAM102", "quantitative-aptitude-mathematics"),
746: ("Philosophy General", "PHL001", "philosophy-general"),
738: ("World General Knowledge", "WGK102", "world-general-knowledge"),
}
# Top-level categories whose posts should map by TOP-LEVEL id (not sub id)
# We resolve: post primary category -> find its top ancestor -> SUBJECT_MAP
# Topic name = the post's direct category name (subcategory), or top-level name if none.
TOP_LEVEL_NAMES = {
1: "General Knowledge MCQs",
37: "World Current Affairs MCQs",
70: "Pakistan Current Affairs MCQs",
48: "Pak Study Mcqs",
39: "Everyday Science Mcqs",
38: "Islamic Studies Mcqs",
50: "Computer Mcqs",
44: "English Mcqs",
45: "Mathematics Mcqs",
41: "Biology Mcqs",
40: "Chemistry Mcqs",
46: "Physics Mcqs",
740: "Medical Mcqs",
79: "Economics Mcqs",
856: "Political Science Mcqs",
521: "English Literature Mcqs",
742: "Psychology Mcqs",
570: "Sociology Mcqs",
109: "Judiciary And Law Mcqs",
982: "International Relations",
54: "Marketing Mcqs",
53: "HRM Mcqs",
52: "Finance Mcqs",
71: "Accounting Mcqs",
80: "Auditing Mcqs",
1025: "Forestry Mcqs",
393: "Agriculture Mcqs",
76: "Statistics Mcqs",
147: "Pedagogy Mcqs",
166: "URDU GENERAL KNOWLEDGE",
320: "CHEMICAL ENGINEERING",
285: "Mechanical Engineering Mcqs",
240: "Civil Engineering Mcqs",
111: "Electrical Engineering Mcqs",
304: "Software Engineering Mcqs",
1066: "Physical Education",
1064: "PAST PAPERS",
737: "ASF ACT 1975 Mcqs",
743: "Election Officer Mcqs",
855: "Pharmaceutical Chemistry Mcqs",
42: "Computer Science",
74: "Management Mcqs",
75: "Business Maths Mcqs",
746: "Philosophy",
738: "Photography Mcqs",
}
def norm(text):
t = text or ""
t = re.sub(r"[^a-z0-9]", "", t.lower())
return t
def slugify(text):
s = re.sub(r"[^a-z0-9\s-]", "", text.lower())
s = re.sub(r"\s+", "-", s)
s = re.sub(r"-+", "-", s).strip("-")
return s
def load_parsed():
items = []
with open(PARSED_FILE, "r", encoding="utf-8") as f:
for ln in f:
try:
items.append(json.loads(ln))
except Exception:
pass
return items
def build_cat_map(cats_file):
cats = {}
with open(cats_file, "r", encoding="utf-8") as f:
for ln in f:
try:
c = json.loads(ln)
cats[c["id"]] = c
except Exception:
pass
return cats
def main():
items = load_parsed()
print("parsed items:", len(items))
cats = build_cat_map(os.path.join(BASE, "data", "pakmcqs_categories.json"))
print("categories:", len(cats))
cat_subject = {} # category_id -> subject_title
cat_topic = {} # category_id -> topic_name
by_subject = {}
for it in items:
cid = it["category_id"]
if cid in cat_subject:
continue
c = cats.get(cid, {})
top_id = cid
guard = 0
cur = c
while cur and cur.get("parent") and guard < 10:
top_id = cur["parent"]
cur = cats.get(top_id)
guard += 1
subj = SUBJECT_MAP.get(top_id)
if not subj:
print("NO SUBJECT MAP for category", cid, c.get("name"), "top:", top_id)
continue
subj_title = subj[0]
topic_name = c.get("name") or TOP_LEVEL_NAMES.get(top_id, "General")
if topic_name == TOP_LEVEL_NAMES.get(top_id):
topic_name = "General " + subj_title
cat_subject[cid] = subj
cat_topic[cid] = topic_name
by_subject.setdefault(subj_title, set()).add(topic_name)
print("\n===== SUBJECT => TOPIC COUNT (planned) =====")
for s, tops in sorted(by_subject.items(), key=lambda x: -len(x[1])):
print(s, "=>", len(tops), "topics")
print("\nunmapped items:", sum(1 for it in items if it["category_id"] not in cat_subject))
if __name__ == "__main__":
main()