File "map_categories.py"

Full path: /home/algopkco/public_html/scraper/map_categories.py
File size: 7.23 B (7.23 KB bytes)
MIME-type: text/x-script.python
Charset: utf-8

Download   Open   Edit   Advanced Editor &nnbsp; Back

import hashlib
import json
import os
import re
import sys
import time

BASE = os.path.dirname(os.path.abspath(__file__))
PARSED_FILE = os.path.join(BASE, "data", "parsed_mcqs.jsonl")

# category_id -> (subject_title, subject_code, subject_slug)
SUBJECT_MAP = {
    1: ("World General Knowledge", "WGK102", "world-general-knowledge"),
    37: ("World Current Affairs", "WCA001", "world-current-affairs"),
    70: ("Pakistan Current Affairs", "PCA001", "pakistan-current-affairs"),
    48: ("Pakistan Study General Test", "psgt11", "pakistan-study-general-test"),
    39: ("Everyday Science General", "EDS001", "everyday-science-general"),
    38: ("Islamic Studies General", "ISG001", "islamic-studies-general"),
    50: ("Computer Science General Test", "CSGT909", "computer-science-general-test"),
    44: ("General English Grammar", "ENGGT01", "general-english-grammar"),
    45: ("Quantitative Aptitude Mathematics", "QAM102", "quantitative-aptitude-mathematics"),
    41: ("Biology General Test", "BioGT", "biology-general-test"),
    40: ("Chemistry  General Test ", "ChemGT", "chemistry-general-test"),
    46: ("Physics  General Test", "phygt", "physics-general-test"),
    740: ("Medical Sciences General", "MDG001", "medical-sciences-general"),
    79: ("Economics General", "ECO001", "economics-general"),
    856: ("Political Science General", "POL001", "political-science-general"),
    521: ("English Literature General", "ELT001", "english-literature-general"),
    742: ("Psychology General", "PSY001", "psychology-general"),
    570: ("Sociology General", "SOC001", "sociology-general"),
    109: ("Judiciary And Law", "JDL001", "judiciary-and-law"),
    982: ("International Relations General", "IRG001", "international-relations-general"),
    54: ("Marketing General", "MKT001", "marketing-general"),
    53: ("HRM General", "HRM001", "hrm-general"),
    52: ("Finance General", "FIN001", "finance-general"),
    71: ("Accounting General", "ACC001", "accounting-general"),
    80: ("Auditing General", "AUD001", "auditing-general"),
    1025: ("Forestry General", "FOR001", "forestry-general"),
    393: ("Agriculture General", "AGR001", "agriculture-general"),
    76: ("Statistics General", "STA001", "statistics-general"),
    147: ("Pedagogy General", "PED001", "pedagogy-general"),
    166: ("Urdu General", "URD001", "urdu-general"),
    320: ("Chemical Engineering General", "CHE001", "chemical-engineering-general"),
    285: ("Mechanical Engineering General", "MEC001", "mechanical-engineering-general"),
    240: ("Civil Engineering General", "CIV001", "civil-engineering-general"),
    111: ("Electrical Engineering General", "ELE001", "electrical-engineering-general"),
    304: ("Software Engineering General", "SWE001", "software-engineering-general"),
    1066: ("Physical Education General", "PHE001", "physical-education-general"),
    1064: ("Past Papers General", "PAP001", "past-papers-general"),
    737: ("Judiciary And Law", "JDL001", "judiciary-and-law"),
    743: ("Political Science General", "POL001", "political-science-general"),
    855: ("Chemistry  General Test ", "ChemGT", "chemistry-general-test"),
    42: ("Computer Science General Test", "CSGT909", "computer-science-general-test"),
    74: ("Management General", "MGT001", "management-general"),
    75: ("Quantitative Aptitude Mathematics", "QAM102", "quantitative-aptitude-mathematics"),
    746: ("Philosophy General", "PHL001", "philosophy-general"),
    738: ("World General Knowledge", "WGK102", "world-general-knowledge"),
}

# Top-level categories whose posts should map by TOP-LEVEL id (not sub id)
# We resolve: post primary category -> find its top ancestor -> SUBJECT_MAP
# Topic name = the post's direct category name (subcategory), or top-level name if none.

TOP_LEVEL_NAMES = {
    1: "General Knowledge MCQs",
    37: "World Current Affairs MCQs",
    70: "Pakistan Current Affairs MCQs",
    48: "Pak Study Mcqs",
    39: "Everyday Science Mcqs",
    38: "Islamic Studies Mcqs",
    50: "Computer Mcqs",
    44: "English Mcqs",
    45: "Mathematics Mcqs",
    41: "Biology Mcqs",
    40: "Chemistry Mcqs",
    46: "Physics Mcqs",
    740: "Medical Mcqs",
    79: "Economics Mcqs",
    856: "Political Science Mcqs",
    521: "English Literature Mcqs",
    742: "Psychology Mcqs",
    570: "Sociology Mcqs",
    109: "Judiciary And Law Mcqs",
    982: "International Relations",
    54: "Marketing Mcqs",
    53: "HRM Mcqs",
    52: "Finance Mcqs",
    71: "Accounting Mcqs",
    80: "Auditing Mcqs",
    1025: "Forestry Mcqs",
    393: "Agriculture Mcqs",
    76: "Statistics Mcqs",
    147: "Pedagogy Mcqs",
    166: "URDU GENERAL KNOWLEDGE",
    320: "CHEMICAL ENGINEERING",
    285: "Mechanical Engineering Mcqs",
    240: "Civil Engineering Mcqs",
    111: "Electrical Engineering Mcqs",
    304: "Software Engineering Mcqs",
    1066: "Physical Education",
    1064: "PAST PAPERS",
    737: "ASF ACT 1975 Mcqs",
    743: "Election Officer Mcqs",
    855: "Pharmaceutical Chemistry Mcqs",
    42: "Computer Science",
    74: "Management Mcqs",
    75: "Business Maths Mcqs",
    746: "Philosophy",
    738: "Photography Mcqs",
}


def norm(text):
    t = text or ""
    t = re.sub(r"[^a-z0-9]", "", t.lower())
    return t


def slugify(text):
    s = re.sub(r"[^a-z0-9\s-]", "", text.lower())
    s = re.sub(r"\s+", "-", s)
    s = re.sub(r"-+", "-", s).strip("-")
    return s


def load_parsed():
    items = []
    with open(PARSED_FILE, "r", encoding="utf-8") as f:
        for ln in f:
            try:
                items.append(json.loads(ln))
            except Exception:
                pass
    return items


def build_cat_map(cats_file):
    cats = {}
    with open(cats_file, "r", encoding="utf-8") as f:
        for ln in f:
            try:
                c = json.loads(ln)
                cats[c["id"]] = c
            except Exception:
                pass
    return cats


def main():
    items = load_parsed()
    print("parsed items:", len(items))
    cats = build_cat_map(os.path.join(BASE, "data", "pakmcqs_categories.json"))
    print("categories:", len(cats))

    cat_subject = {}   # category_id -> subject_title
    cat_topic = {}     # category_id -> topic_name
    by_subject = {}

    for it in items:
        cid = it["category_id"]
        if cid in cat_subject:
            continue
        c = cats.get(cid, {})
        top_id = cid
        guard = 0
        cur = c
        while cur and cur.get("parent") and guard < 10:
            top_id = cur["parent"]
            cur = cats.get(top_id)
            guard += 1
        subj = SUBJECT_MAP.get(top_id)
        if not subj:
            print("NO SUBJECT MAP for category", cid, c.get("name"), "top:", top_id)
            continue
        subj_title = subj[0]
        topic_name = c.get("name") or TOP_LEVEL_NAMES.get(top_id, "General")
        if topic_name == TOP_LEVEL_NAMES.get(top_id):
            topic_name = "General " + subj_title
        cat_subject[cid] = subj
        cat_topic[cid] = topic_name
        by_subject.setdefault(subj_title, set()).add(topic_name)

    print("\n===== SUBJECT => TOPIC COUNT (planned) =====")
    for s, tops in sorted(by_subject.items(), key=lambda x: -len(x[1])):
        print(s, "=>", len(tops), "topics")
    print("\nunmapped items:", sum(1 for it in items if it["category_id"] not in cat_subject))


if __name__ == "__main__":
    main()