paramj/topic_modelling
0
1"""tools.py — Sentence-level BERTopic pipeline + Mistral LLM. Version 3.0.0 | 4 April 2026. ZERO for/while/if.2 3PIPELINE:4 Paper → split into sentences → each sentence gets paper_id + sent_id + metadata5 → embed sentences (384d) → AgglomerativeClustering cosine → centroid nearest 5 sentences6 → Mistral labels topics from sentence evidence + paper metadata7 → one paper can span MULTIPLE topics8"""9from langchain_core.tools import tool10import os11import json12import re13import numpy as np14import pandas as pd15 16# ═══════════════════════════════════════════════17# DEBUG + STATE + CONSTANTS18# ═══════════════════════════════════════════════19DEBUG = True20debug = {True: print, False: lambda *a, **k: None}[DEBUG]21 22CHECKPOINT_DIR = "/tmp/checkpoints"23os.makedirs(CHECKPOINT_DIR, exist_ok=True)24 25NEAREST_K = 526SENT_SPLIT_RE = r'(?<=[.!?])\s+(?=[A-Z])'27MIN_SENT_LEN = 3028 29RUN_CONFIGS = {30 "abstract": ["Abstract"],31 "title": ["Title"],32}33 34_data = {}35 36 37# ═══════════════════════════════════════════════38# HELPER: Split text into sentences (regex, no nltk)39# ═══════════════════════════════════════════════40def _split_sentences(text):41 """Split text on sentence boundaries. Filters short fragments (<30 chars).42 Uses regex: split after .!? followed by uppercase letter."""43 raw = re.split(SENT_SPLIT_RE, str(text))44 return list(filter(lambda s: len(s.strip()) >= MIN_SENT_LEN, raw))45 46 47# ═══════════════════════════════════════════════48# TOOL 1: Load Scopus CSV49# ═══════════════════════════════════════════════50@tool51def load_scopus_csv(filepath: str) -> str:52 """Load a Scopus CSV export and show preview. Call this first.53 54 Args:55 filepath: Path to the uploaded .csv file.56 57 Returns:58 Row count, column names, and sample data."""59 debug(f"\n>>> TOOL: load_scopus_csv(filepath='{filepath}')")60 df = pd.read_csv(filepath, encoding="utf-8-sig")61 _data["df"] = df62 debug(f">>> Loaded {len(df)} rows, {len(df.columns)} columns")63 target_cols = list(filter(lambda c: c in df.columns, ["Title", "Abstract", "Author Keywords"]))64 sample = df[target_cols].head(3).to_string(max_colwidth=80)65 null_counts = ", ".join(list(map(66 lambda c: f"{c}: {df[c].notna().sum()}/{len(df)}", target_cols)))67 68 # Estimate sentence counts69 sample_sents = df["Abstract"].head(5).apply(_split_sentences).apply(len)70 avg_abstract_sents = sample_sents.mean()71 est_abstract = int(avg_abstract_sents * len(df))72 title_count = int(df["Title"].notna().sum())73 74 return (f"📊 **Dataset Statistics:**\n"75 f"- **Papers:** {len(df)}\n"76 f"- **Abstract sentences:** ~{est_abstract} (~{avg_abstract_sents:.0f} per paper)\n"77 f"- **Title sentences:** {title_count} (1 per paper)\n"78 f"- **Non-null:** {null_counts}\n\n"79 f"Columns: {', '.join(list(df.columns)[:15])}\n\n"80 f"Sample:\n{sample}")81 82 83# ═══════════════════════════════════════════════84# TOOL 2: Sentence-Level BERTopic Pipeline85# ═══════════════════════════════════════════════86@tool87def run_bertopic_discovery(run_key: str, threshold: float = 0.7) -> str:88 """Sentence-level BERTopic: split papers → embed sentences → cosine similarity clustering → centroid nearest 5 → Plotly charts.89 Each sentence keeps paper_id, sent_id, and metadata. One paper can span multiple topics.90 Uses AgglomerativeClustering with cosine distance — groups sentences by similarity threshold.91 92 Args:93 run_key: One of 'abstract' or 'title' — selects which columns to split into sentences.94 threshold: Cosine distance threshold (0.0-1.0). Lower = stricter = more topics.95 0.5 = very strict (~2000 topics), 0.7 = recommended (~100 topics, default), 0.8 = loose (~30 topics), 0.9 = very loose (~10 topics).96 97 Returns:98 Topic summary with sentence counts, paper counts, and 5 nearest centroid sentences."""99 debug(f"\n>>> TOOL: run_bertopic_discovery(run_key='{run_key}', threshold={threshold})")100 from bertopic import BERTopic101 from sentence_transformers import SentenceTransformer102 103 df = _data["df"].copy()104 cols = RUN_CONFIGS[run_key]105 available = list(filter(lambda c: c in df.columns, cols))106 debug(f">>> Columns: {available}")107 108 # ── Step 1: Assemble text per paper ──109 df["_text"] = df[available].fillna("").agg(" ".join, axis=1)110 df["_paper_id"] = df.index111 debug(f">>> {len(df)} papers assembled")112 113 # ── Step 2: Split into sentences — regex, no nltk ──114 debug(">>> Splitting into sentences...")115 df["_sentences"] = df["_text"].apply(_split_sentences)116 debug(f">>> Sentence counts: min={df['_sentences'].apply(len).min()}, "117 f"max={df['_sentences'].apply(len).max()}, "118 f"mean={df['_sentences'].apply(len).mean():.1f}")119 120 # ── Step 3: Explode to sentence-level DataFrame ──121 meta_cols = ["_paper_id", "Title", "Author Keywords", "_sentences"]122 available_meta = list(filter(lambda c: c in df.columns, meta_cols))123 sent_df = df[available_meta].explode("_sentences").rename(124 columns={"_sentences": "text"}).reset_index(drop=True)125 sent_df = sent_df.dropna(subset=["text"]).reset_index(drop=True)126 sent_df["sent_id"] = sent_df.groupby("_paper_id").cumcount()127 128 # ── Step 3b: Filter out publisher boilerplate sentences ──129 # Scopus abstracts contain copyright/license noise that clustering picks up as topics.130 # These are NOT research content — remove before embedding.131 debug(">>> Filtering publisher boilerplate...")132 _n_before = len(sent_df)133 boilerplate_patterns = "|".join([134 r"Licensee MDPI",135 r"Published by Informa",136 r"Published by Elsevier",137 r"Taylor & Francis",138 r"Copyright ©",139 r"Creative Commons",140 r"open access article",141 r"Inderscience Enterprises",142 r"All rights reserved",143 r"This is an open access",144 r"distributed under the terms",145 r"The Author\(s\)",146 r"Springer Nature",147 r"Emerald Publishing",148 r"limitations and future",149 r"limitations and implications",150 r"limitations are discussed",151 r"limitations have been discussed",152 r"implications are discussed",153 r"implications were discussed",154 r"implications are presented",155 r"concludes with .* implications",156 ])157 clean_mask = ~sent_df["text"].str.contains(boilerplate_patterns, case=False, regex=True, na=False)158 sent_df = sent_df[clean_mask].reset_index(drop=True)159 sent_df["sent_id"] = sent_df.groupby("_paper_id").cumcount()160 debug(f">>> Filtered: {_n_before} → {len(sent_df)} sentences ({_n_before - len(sent_df)} boilerplate removed)")161 n_sentences = len(sent_df)162 n_papers = len(df)163 debug(f">>> {n_sentences} sentences from {n_papers} papers")164 165 # ── Step 4: Embed sentences (384d, L2-normalized) ──166 # BERTopic FAQ: "normalize them first to force a cosine-related distance metric"167 # Math: for L2-normalized vectors, euclidean²(a,b) = 2(1 - cos(a,b)) → same clusters as cosine168 debug(">>> Embedding sentences with all-MiniLM-L6-v2 (L2-normalized)...")169 docs = sent_df["text"].tolist()170 embedder = SentenceTransformer("all-MiniLM-L6-v2")171 embeddings = embedder.encode(docs, show_progress_bar=False, normalize_embeddings=True)172 debug(f">>> Embeddings: {embeddings.shape}, normalized: True")173 174 # Save checkpoint175 np.save(f"{CHECKPOINT_DIR}/rq4_{run_key}_emb.npy", embeddings)176 177 # ── Step 5: Agglomerative Clustering with COSINE similarity threshold ──178 # Groups sentences where cosine_distance < threshold → same cluster179 # No dimension reduction. No density estimation. Pure similarity grouping.180 debug(f">>> AgglomerativeClustering cosine threshold={threshold} on 384d embeddings...")181 from sklearn.preprocessing import FunctionTransformer182 from sklearn.cluster import AgglomerativeClustering183 no_umap = FunctionTransformer()184 cluster_model = AgglomerativeClustering(185 n_clusters=None,186 metric="cosine",187 linkage="average",188 distance_threshold=threshold,189 )190 topic_model = BERTopic(191 hdbscan_model=cluster_model,192 umap_model=no_umap,193 )194 topics, probs = topic_model.fit_transform(docs, embeddings)195 n_topics = len(set(topics)) - int(-1 in topics)196 n_outliers = int(np.sum(np.array(topics) == -1))197 debug(f">>> {n_topics} topics, {n_outliers} outlier sentences")198 199 # Store for later tools200 _data[f"{run_key}_model"] = topic_model201 _data[f"{run_key}_topics"] = np.array(topics)202 _data[f"{run_key}_embeddings"] = embeddings203 _data[f"{run_key}_sent_df"] = sent_df204 205 # ── Step 6: BERTopic Plotly visualizations (skip charts that need 3+ topics) ──206 debug(f">>> Generating visualizations ({n_topics} topics)...")207 # visualize_topics() uses UMAP internally → crashes with < 3 topics208 (n_topics >= 3) and topic_model.visualize_topics().write_html(209 f"/tmp/rq4_{run_key}_intertopic.html", include_plotlyjs="cdn")210 # barchart works with 1+ topics211 (n_topics >= 1) and topic_model.visualize_barchart(212 top_n_topics=min(10, max(1, n_topics))).write_html(213 f"/tmp/rq4_{run_key}_bars.html", include_plotlyjs="cdn")214 # hierarchy needs 2+ topics215 (n_topics >= 2) and topic_model.visualize_hierarchy().write_html(216 f"/tmp/rq4_{run_key}_hierarchy.html", include_plotlyjs="cdn")217 # heatmap needs 2+ topics218 (n_topics >= 2) and topic_model.visualize_heatmap().write_html(219 f"/tmp/rq4_{run_key}_heatmap.html", include_plotlyjs="cdn")220 debug(f">>> Visualizations saved (skipped charts needing more topics)")221 222 # ── Step 7: Centroid nearest 5 SENTENCES — COSINE similarity ──223 topics_arr = np.array(topics)224 topic_info = topic_model.get_topic_info()225 valid_rows = list(filter(lambda r: r["Topic"] != -1, topic_info.to_dict("records")))226 227 def _centroid_nearest(row):228 """Find 5 sentences nearest to topic centroid via cosine similarity."""229 mask = topics_arr == row["Topic"]230 member_idx = np.where(mask)[0]231 member_embs = embeddings[mask]232 centroid = member_embs.mean(axis=0)233 # Cosine distance: 1 - cos_sim. For normalized vectors: cos_sim = dot product234 norms = np.linalg.norm(member_embs, axis=1) * np.linalg.norm(centroid)235 cosine_sim = (member_embs @ centroid) / (norms + 1e-10)236 dists = 1 - cosine_sim237 nearest = np.argsort(dists)[:NEAREST_K]238 239 # 5 nearest sentences with paper metadata240 nearest_evidence = list(map(lambda i: {241 "sentence": str(sent_df.iloc[member_idx[i]]["text"])[:250],242 "paper_id": int(sent_df.iloc[member_idx[i]]["_paper_id"]),243 "title": str(sent_df.iloc[member_idx[i]].get("Title", ""))[:150],244 "keywords": str(sent_df.iloc[member_idx[i]].get("Author Keywords", ""))[:150],245 }, nearest))246 247 # Count unique papers in this topic + collect their titles248 topic_papers_df = sent_df.iloc[member_idx].drop_duplicates(subset=["_paper_id"])249 unique_papers = len(topic_papers_df)250 paper_titles = list(map(251 lambda idx: str(topic_papers_df.iloc[idx].get("Title", ""))[:200],252 range(min(50, unique_papers)))) # cap at 50 titles per topic253 254 return {"topic_id": int(row["Topic"]),255 "sentence_count": int(row["Count"]),256 "paper_count": int(unique_papers),257 "top_words": str(row.get("Name", ""))[:100],258 "nearest": nearest_evidence,259 "paper_titles": paper_titles}260 261 summaries = list(map(_centroid_nearest, valid_rows))262 json.dump(summaries, open(f"{CHECKPOINT_DIR}/rq4_{run_key}_summaries.json", "w"), indent=2, default=str)263 debug(f">>> {len(summaries)} topics saved ({NEAREST_K} nearest sentences each)")264 265 # ── Format output ──266 lines = list(map(267 lambda s: f" Topic {s['topic_id']} ({s['sentence_count']} sentences, {s['paper_count']} papers): {s['top_words']}",268 summaries))269 return (f"[{run_key}] {n_topics} topics from {n_sentences} sentences ({n_papers} papers, {n_outliers} outliers).\n\n"270 + "\n".join(lines)271 + f"\n\nVisualizations: /tmp/rq4_{run_key}_*.html (4 files)"272 + f"\nCheckpoints: {CHECKPOINT_DIR}/rq4_{run_key}_emb.npy + summaries.json")273 274 275# ═══════════════════════════════════════════════276# TOOL 3: Label Topics with Mistral (sentence evidence)277# ═══════════════════════════════════════════════278@tool279def label_topics_with_llm(run_key: str) -> str:280 """Send 5 nearest centroid sentences + paper metadata to Mistral for labeling.281 Each sentence shows which paper it came from (title + keywords).282 283 Args:284 run_key: One of 'abstract' or 'title'.285 286 Returns:287 Labeled topics with sentence-level evidence."""288 debug(f"\n>>> TOOL: label_topics_with_llm(run_key='{run_key}')")289 from langchain_mistralai import ChatMistralAI290 from langchain_core.prompts import PromptTemplate291 from langchain_core.output_parsers import JsonOutputParser292 293 summaries = json.load(open(f"{CHECKPOINT_DIR}/rq4_{run_key}_summaries.json"))294 debug(f">>> Loaded {len(summaries)} topics ({NEAREST_K} sentences each)")295 296 # Limit to top 50 largest topics — prevents Mistral rate limit on 2000+ topics297 MAX_LABEL_TOPICS = 100298 sorted_summaries = sorted(summaries, key=lambda s: s.get("sentence_count", 0), reverse=True)299 summaries_to_label = sorted_summaries[:MAX_LABEL_TOPICS]300 skipped = max(0, len(summaries) - MAX_LABEL_TOPICS)301 debug(f">>> Labeling top {len(summaries_to_label)} topics (skipped {skipped} small clusters)")302 303 # Format all topics — show sentence + paper metadata as evidence304 topics_block = "\n\n".join(list(map(305 lambda s: (f"Topic {s['topic_id']} ({s['sentence_count']} sentences from {s['paper_count']} papers):\n"306 f" Top words: {s['top_words']}\n"307 f" {NEAREST_K} nearest centroid sentences:\n"308 + "\n".join(list(map(309 lambda e: (f" - \"{e['sentence'][:200]}\"\n"310 f" Paper: \"{e['title']}\"\n"311 f" Keywords: {e['keywords']}"),312 s["nearest"])))),313 summaries_to_label)))314 315 prompt = PromptTemplate.from_template(316 "You are a research topic classifier for academic papers about Technology and Tourism.\n\n"317 "For EACH topic below, you are given the 5 sentences nearest to the topic centroid,\n"318 "plus the paper title and author keywords each sentence came from.\n\n"319 "Return a JSON ARRAY with one object per topic:\n"320 "- topic_id: integer\n"321 "- label: short descriptive name (3-6 words, specific — NOT generic like 'tourism studies')\n"322 "- category: general research area (e.g., 'technology adoption', 'consumer behavior',\n"323 " 'virtual reality', 'social media marketing', 'sustainability', 'cultural heritage',\n"324 " 'AI and machine learning', 'online reviews', 'destination marketing',\n"325 " 'tourist psychology', 'hotel management', 'sharing economy',\n"326 " 'mobile applications', 'research methodology', 'data analytics')\n"327 " DO NOT use PACIS/ICIS categories — just plain descriptive research area.\n"328 "- confidence: high, medium, or low\n"329 "- reasoning: 1 sentence explaining WHY you chose this label based on the evidence sentences\n"330 "- niche: true or false (true = very specific sub-area with <20 sentences)\n\n"331 "CRITICAL: be SPECIFIC in labels. Do NOT use broad terms.\n"332 "Return ONLY valid JSON array, no markdown.\n\n"333 "Topics:\n{topics}")334 335 llm = ChatMistralAI(model="mistral-small-latest", temperature=0, timeout=300)336 chain = prompt | llm | JsonOutputParser()337 debug(">>> Calling Mistral (single call, all topics)...")338 labels = chain.invoke({"topics": topics_block})339 debug(f">>> Got {len(labels)} labels")340 341 # Merge labels with summaries342 labeled = list(map(lambda pair: {**pair[0], **pair[1]},343 zip(summaries, (labels + summaries)[:len(summaries)])))344 json.dump(labeled, open(f"{CHECKPOINT_DIR}/rq4_{run_key}_labels.json", "w"), indent=2, default=str)345 debug(f">>> Labels saved: {CHECKPOINT_DIR}/rq4_{run_key}_labels.json")346 347 # Format — show label + evidence sentences + paper source348 lines = list(map(349 lambda l: (f" **Topic {l.get('topic_id', '?')}: {l.get('label', '?')}** "350 f"[{l.get('category', '?')}] conf={l.get('confidence', '?')} "351 f"({l.get('sentence_count', 0)} sentences, {l.get('paper_count', 0)} papers)\n"352 + "\n".join(list(map(353 lambda e: f" → \"{e['sentence'][:120]}...\" — _{e['title'][:60]}_",354 l.get("nearest", []))))),355 labeled))356 return f"[{run_key}] {len(labeled)} topics labeled by Mistral:\n\n" + "\n\n".join(lines)357 358 359# ═══════════════════════════════════════════════360# TOOL 4: Generate Comparison Table361# ═══════════════════════════════════════════════362@tool363def generate_comparison_csv() -> str:364 """Compare Mistral-labeled topics across completed runs. Includes sentence + paper counts.365 366 Returns:367 Comparison table + CSV path."""368 debug(f"\n>>> TOOL: generate_comparison_csv()")369 completed = list(filter(370 lambda k: os.path.exists(f"{CHECKPOINT_DIR}/rq4_{k}_labels.json"), RUN_CONFIGS.keys()))371 debug(f">>> Completed runs: {completed}")372 373 def _load_run(run_key):374 labels = json.load(open(f"{CHECKPOINT_DIR}/rq4_{run_key}_labels.json"))375 return list(map(lambda l: {376 "run": run_key, "topic_id": l.get("topic_id", ""),377 "label": l.get("label", ""), "category": l.get("category", ""),378 "confidence": l.get("confidence", ""), "niche": l.get("niche", ""),379 "sentences": l.get("sentence_count", 0),380 "papers": l.get("paper_count", 0),381 "top_words": l.get("top_words", ""),382 }, labels))383 384 all_rows = sum(list(map(_load_run, completed)), [])385 df = pd.DataFrame(all_rows)386 path = "/tmp/rq4_comparison.csv"387 df.to_csv(path, index=False)388 debug(f">>> Comparison CSV: {path} ({len(df)} rows)")389 return f"Comparison saved: {path} ({len(completed)} runs, {len(df)} topics)\n\n{df.to_string(index=False)}"390 391 392# ═══════════════════════════════════════════════393# TOOL 5: Export 500-Word Narrative394# ═══════════════════════════════════════════════395@tool396def export_narrative(run_key: str) -> str:397 """Generate 500-word narrative for research paper Section 7 via Mistral.398 399 Args:400 run_key: One of 'abstract' or 'title'.401 402 Returns:403 500-word narrative + save path."""404 debug(f"\n>>> TOOL: export_narrative(run_key='{run_key}')")405 from langchain_mistralai import ChatMistralAI406 407 labels = json.load(open(f"{CHECKPOINT_DIR}/rq4_{run_key}_labels.json"))408 topics_text = "\n".join(list(map(409 lambda l: f"- {l.get('label', '?')} ({l.get('sentence_count', 0)} sentences from "410 f"{l.get('paper_count', 0)} papers, category: {l.get('category', '?')}, "411 f"confidence: {l.get('confidence', '?')}, niche: {l.get('niche', '?')})",412 labels)))413 414 llm = ChatMistralAI(model="mistral-small-latest", temperature=0.3, timeout=300)415 result = llm.invoke(416 f"Write exactly 500 words for a research paper Section 7 titled "417 f"'Topic Modeling Results — BERTopic Discovery'.\n\n"418 f"Dataset: 1390 Scopus papers on Tourism and AI.\n"419 f"Method: Sentence-level BERTopic — each abstract split into sentences,\n"420 f"embedded with all-MiniLM-L6-v2 (384d), clustered with AgglomerativeClustering (cosine).\n"421 f"Note: One paper can contribute sentences to MULTIPLE topics.\n"422 f"Run config: '{run_key}' columns.\n\n"423 f"Topics discovered:\n{topics_text}\n\n"424 f"Include: methodology justification for sentence-level approach,\n"425 f"key themes, emerging niches, limitations, future work.")426 427 path = "/tmp/rq4_narrative.txt"428 open(path, "w", encoding="utf-8").write(result.content)429 debug(f">>> Narrative saved: {path} ({len(result.content)} chars)")430 return f"Narrative saved: {path}\n\n{result.content}"431 432 433# ═══════════════════════════════════════════════434# TOOL 6: Consolidate Round 1 Topics into Themes435# ═══════════════════════════════════════════════436@tool437def consolidate_into_themes(run_key: str, theme_map: dict) -> str:438 """ROUND 2: Merge fine-grained Round 1 topics into broader themes.439 Researcher decides which topics to group. Recomputes centroids and evidence.440 441 Args:442 run_key: 'abstract' or 'title'.443 theme_map: Dict mapping theme names to topic ID lists.444 Example: {"AI in Tourism": [0, 1, 5], "VR Tourism": [2, 3]}445 446 Returns:447 Consolidated themes with new 5-nearest sentence evidence per theme."""448 debug(f"\n>>> TOOL: consolidate_into_themes(run_key='{run_key}', {len(theme_map)} themes)")449 450 topics_arr = _data[f"{run_key}_topics"]451 embeddings = _data[f"{run_key}_embeddings"]452 sent_df = _data[f"{run_key}_sent_df"]453 454 def _build_theme(item):455 """Merge listed topics into one theme. Recompute centroid + 5 nearest."""456 theme_name, topic_ids = item457 mask = np.isin(topics_arr, topic_ids)458 member_idx = np.where(mask)[0]459 member_embs = embeddings[mask]460 centroid = member_embs.mean(axis=0)461 norms = np.linalg.norm(member_embs, axis=1) * np.linalg.norm(centroid)462 cosine_sim = (member_embs @ centroid) / (norms + 1e-10)463 dists = 1 - cosine_sim464 nearest = np.argsort(dists)[:NEAREST_K]465 466 nearest_evidence = list(map(lambda i: {467 "sentence": str(sent_df.iloc[member_idx[i]]["text"])[:250],468 "paper_id": int(sent_df.iloc[member_idx[i]]["_paper_id"]),469 "title": str(sent_df.iloc[member_idx[i]].get("Title", ""))[:150],470 "keywords": str(sent_df.iloc[member_idx[i]].get("Author Keywords", ""))[:150],471 }, nearest))472 473 unique_papers = sent_df.iloc[member_idx]["_paper_id"].nunique()474 475 # Collect paper titles (up to 50)476 topic_papers_df = sent_df.iloc[member_idx].drop_duplicates(subset=["_paper_id"])477 paper_titles = list(map(478 lambda idx: str(topic_papers_df.iloc[idx].get("Title", ""))[:200],479 range(min(50, len(topic_papers_df)))))480 481 return {"label": theme_name, "merged_topics": list(topic_ids),482 "sentence_count": int(mask.sum()), "paper_count": int(unique_papers),483 "nearest": nearest_evidence, "paper_titles": paper_titles}484 485 # Add topic_id to each theme (sequential)486 themes_raw = list(map(_build_theme, theme_map.items()))487 themes = list(map(488 lambda pair: {**pair[1], "topic_id": pair[0]},489 enumerate(themes_raw)))490 json.dump(themes, open(f"{CHECKPOINT_DIR}/rq4_{run_key}_themes.json", "w"), indent=2, default=str)491 debug(f">>> {len(themes)} themes saved: {CHECKPOINT_DIR}/rq4_{run_key}_themes.json")492 493 # Format — show theme + merged topics + evidence494 lines = list(map(495 lambda t: (f" **{t['label']}** ({t['sentence_count']} sentences, {t['paper_count']} papers)\n"496 f" Merged from topics: {t['merged_topics']}\n"497 f" Evidence:\n"498 + "\n".join(list(map(499 lambda e: f" → \"{e['sentence'][:120]}...\" — _{e['title'][:60]}_",500 t["nearest"])))),501 themes))502 return f"[{run_key}] Round 2: {len(themes)} themes consolidated:\n\n" + "\n\n".join(lines)503 504 505# ═══════════════════════════════════════════════506# TOOL 7: Compare Themes with PAJAIS Taxonomy507# ═══════════════════════════════════════════════508 509# Established IS topic taxonomy from:510# Jiang, Liang & Tsai (2019) "Knowledge Profile in PAJAIS"511# Pacific Asia Journal of the AIS, 11(1), 1-24. doi:10.17705/1pais.11101512PAJAIS_TAXONOMY = [513 "Electronic and Mobile Business / Social Commerce",514 "Human Behavior and IS / Human-Computer Interaction",515 "IS/IT Strategy, Leadership, Governance",516 "Business Intelligence and Data Analytics",517 "Design Science and IS",518 "Enterprise Systems and BPM",519 "IS Implementation, Adoption, and Diffusion",520 "Social Media and Business Impact",521 "Cultural and Global Issues in IS",522 "IS Security and Privacy",523 "IS Smart / IoT",524 "Knowledge Management",525 "ICT / Digital Platform / IT and Work",526 "IS Healthcare",527 "IT Project Management",528 "Service Science and IS",529 "Social and Organizational Aspects of IS",530 "Research Methods and Philosophy",531 "E-Finance / Economics of IS",532 "E-Government",533 "IS Education and Learning",534 "Green IT and Sustainability",535]536 537 538@tool539def compare_with_taxonomy(run_key: str) -> str:540 """Compare BERTopic themes against established PAJAIS/PACIS taxonomy541 (Jiang, Liang & Tsai, 2019). Identifies which themes map to known542 categories and which are NOVEL/EMERGING (not in existing taxonomy).543 Researcher reviews mapping and approves new theme consolidation.544 545 Args:546 run_key: 'abstract' or 'title'.547 548 Returns:549 Mapping table: BERTopic theme → PAJAIS category (or NOVEL)."""550 debug(f"\n>>> TOOL: compare_with_taxonomy(run_key='{run_key}')")551 from langchain_mistralai import ChatMistralAI552 from langchain_core.prompts import PromptTemplate553 from langchain_core.output_parsers import JsonOutputParser554 555 # Load themes (prefer consolidated themes, fall back to labels)556 themes_path = f"{CHECKPOINT_DIR}/rq4_{run_key}_themes.json"557 labels_path = f"{CHECKPOINT_DIR}/rq4_{run_key}_labels.json"558 source_path = (os.path.exists(themes_path) and themes_path) or labels_path559 themes = json.load(open(source_path))560 debug(f">>> Loaded {len(themes)} themes from {source_path}")561 562 # Format themes for Mistral563 themes_text = "\n".join(list(map(564 lambda t: f"- {t.get('label', '?')} "565 f"({t.get('paper_count', t.get('count', '?'))} papers)",566 themes)))567 568 taxonomy_text = "\n".join(list(map(lambda c: f"- {c}", PAJAIS_TAXONOMY)))569 570 prompt = PromptTemplate.from_template(571 "You are an IS research taxonomy expert.\n\n"572 "Compare each BERTopic theme against the established PAJAIS/PACIS taxonomy.\n"573 "For EACH theme, return a JSON ARRAY with:\n"574 "- label: the BERTopic theme name\n"575 "- pajais_match: closest PAJAIS category (or 'NOVEL' if no match)\n"576 "- match_confidence: high, medium, low, or none\n"577 "- reasoning: why this mapping (1 sentence)\n"578 "- is_novel: true if this theme represents an emerging area not in the taxonomy\n\n"579 "Return ONLY valid JSON array.\n\n"580 "BERTopic Themes:\n{themes}\n\n"581 "PAJAIS Taxonomy (Jiang et al., 2019):\n{taxonomy}")582 583 llm = ChatMistralAI(model="mistral-small-latest", temperature=0, timeout=300)584 chain = prompt | llm | JsonOutputParser()585 debug(">>> Calling Mistral for taxonomy comparison...")586 mappings = chain.invoke({"themes": themes_text, "taxonomy": taxonomy_text})587 debug(f">>> Got {len(mappings)} mappings")588 589 # Save mapping590 json.dump(mappings, open(f"{CHECKPOINT_DIR}/rq4_{run_key}_taxonomy_map.json", "w"), indent=2, default=str)591 592 # Count novel vs mapped593 novel = list(filter(lambda m: m.get("is_novel", False), mappings))594 mapped = list(filter(lambda m: not m.get("is_novel", False), mappings))595 596 # Format output597 mapped_lines = list(map(598 lambda m: f" ✅ {m.get('label', '?')} → **{m.get('pajais_match', '?')}** "599 f"(conf={m.get('match_confidence', '?')}) _{m.get('reasoning', '')}_",600 mapped))601 novel_lines = list(map(602 lambda m: f" 🆕 **{m.get('label', '?')}** → NOVEL "603 f"_{m.get('reasoning', '')}_",604 novel))605 606 return (f"[{run_key}] Taxonomy comparison (Jiang et al., 2019):\n\n"607 f"**Mapped to PAJAIS categories ({len(mapped)}):**\n" + "\n".join(mapped_lines) +608 f"\n\n**NOVEL / Emerging themes ({len(novel)}):**\n" + "\n".join(novel_lines) +609 f"\n\nSaved: {CHECKPOINT_DIR}/rq4_{run_key}_taxonomy_map.json")610 611 612# ═══════════════════════════════════════════════613# GET ALL TOOLS614# ═══════════════════════════════════════════════615def get_all_tools():616 """Return all 7 tools with error handling enabled."""617 tools = [load_scopus_csv, run_bertopic_discovery, label_topics_with_llm,618 consolidate_into_themes, compare_with_taxonomy,619 generate_comparison_csv, export_narrative]620 list(map(lambda t: setattr(t, 'handle_tool_error', True), tools))621 debug(f">>> tools.py: {len(tools)} tools ready (handle_tool_error=True)")622 list(map(lambda t: debug(f">>> - {t.name}"), tools))623 return tools624 