314 lines
12 KiB
Python
314 lines
12 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Penarik data penelitian dosen via SerpAPI (Google Scholar) + kuartil Scopus dari SINTA.
|
|
|
|
Dua langkah:
|
|
1. publications : tarik SEMUA publikasi tiap dosen dari Google Scholar lewat
|
|
SerpAPI (engine google_scholar_author, 100 artikel/request) berdasarkan
|
|
kolom dosen.scholar, lalu insert baris baru ke tabel `penelitian`.
|
|
Kolom `link` diisi citation_id (format sama dgn data lama).
|
|
2. scopus : tarik daftar publikasi Scopus tiap dosen dari SINTA
|
|
(view=scopus, ada label kuartil), cocokkan judul, lalu UPDATE kolom
|
|
`penelitian.scopus` (Q1..Q4).
|
|
|
|
Konfigurasi environment:
|
|
DB_HOST, DB_USER, DB_PASS, DB_NAME
|
|
SERPAPI_KEY (fallback: file apikey.json di root project)
|
|
SERPAPI_MAX_REQUESTS (default 90 - pengaman kuota free tier 100/bulan)
|
|
|
|
CLI:
|
|
python scholar_api_scraper.py --dry-run # tanpa tulis DB
|
|
python scholar_api_scraper.py # publications + scopus
|
|
python scholar_api_scraper.py --only publications
|
|
python scholar_api_scraper.py --only scopus # tidak pakai kuota SerpAPI
|
|
"""
|
|
|
|
import argparse
|
|
import logging
|
|
import os
|
|
import re
|
|
|
|
import requests
|
|
|
|
try: # di image Airflow, modul berada di /opt/airflow/scripts (PYTHONPATH=/opt/airflow)
|
|
from scripts.sinta_scraper import (
|
|
db_connect,
|
|
ensure_schema,
|
|
get_author_publications,
|
|
get_department_authors,
|
|
make_session,
|
|
match_dosen,
|
|
norm_title,
|
|
DEFAULT_DEPT_URL,
|
|
)
|
|
except ImportError:
|
|
from sinta_scraper import (
|
|
db_connect,
|
|
ensure_schema,
|
|
get_author_publications,
|
|
get_department_authors,
|
|
make_session,
|
|
match_dosen,
|
|
norm_title,
|
|
DEFAULT_DEPT_URL,
|
|
)
|
|
|
|
log = logging.getLogger("scholar_api")
|
|
SERPAPI_URL = "https://serpapi.com/search.json"
|
|
|
|
|
|
def get_api_key():
|
|
key = os.getenv("SERPAPI_KEY")
|
|
if key:
|
|
return key.strip()
|
|
path = os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "apikey.json")
|
|
if os.path.exists(path):
|
|
return open(path, encoding="utf-8").read().strip().strip('"')
|
|
raise RuntimeError("SERPAPI_KEY tidak di-set dan apikey.json tidak ditemukan")
|
|
|
|
|
|
def load_dosen():
|
|
"""Daftar dosen dari tabel dosen. Fallback ke halaman departemen SINTA
|
|
(ID Scholar dari URL foto) bila DB tidak bisa diakses (mis. saat dry-run)."""
|
|
try:
|
|
conn = db_connect()
|
|
with conn.cursor() as cur:
|
|
cur.execute("SELECT id_dosen, nama, scholar, sinta FROM dosen")
|
|
rows = [
|
|
{
|
|
"id_dosen": r[0],
|
|
"nama": re.sub(r"\s+", " ", (r[1] or "")).strip(),
|
|
"scholar": (r[2] or "").strip(),
|
|
"sinta": (r[3] or "").strip() if r[3] else "",
|
|
}
|
|
for r in cur.fetchall()
|
|
]
|
|
conn.close()
|
|
return rows, True
|
|
except Exception as e:
|
|
log.warning("DB tidak bisa diakses (%s) - fallback ke daftar SINTA", e)
|
|
authors = get_department_authors(make_session(), DEFAULT_DEPT_URL)
|
|
return [
|
|
{"id_dosen": None, "nama": a["nama"], "scholar": a["gs_id"] or "", "sinta": a["sinta_id"]}
|
|
for a in authors
|
|
], False
|
|
|
|
|
|
# ------------------------------------------------- langkah 1: SerpAPI Scholar
|
|
|
|
def fetch_author_articles(key, author_id, budget):
|
|
"""Semua artikel satu author dari SerpAPI. Return (nama_profil_gs, list artikel)."""
|
|
articles, start, profile_name = [], 0, ""
|
|
while True:
|
|
if budget["used"] >= budget["max"]:
|
|
log.warning("Budget SerpAPI (%d request) habis - stop di %s", budget["max"], author_id)
|
|
break
|
|
resp = requests.get(
|
|
SERPAPI_URL,
|
|
params={
|
|
"engine": "google_scholar_author",
|
|
"author_id": author_id,
|
|
"num": 100,
|
|
"start": start,
|
|
"api_key": key,
|
|
"hl": "id",
|
|
},
|
|
timeout=60,
|
|
)
|
|
budget["used"] += 1
|
|
data = resp.json()
|
|
if "error" in data:
|
|
raise RuntimeError(data["error"])
|
|
profile_name = data.get("author", {}).get("name", profile_name)
|
|
batch = data.get("articles", [])
|
|
articles.extend(batch)
|
|
if len(batch) < 100 or "next" not in data.get("serpapi_pagination", {}):
|
|
break
|
|
start += 100
|
|
return profile_name, articles
|
|
|
|
|
|
def _apply_filter(rows, dosen_filter):
|
|
if not dosen_filter:
|
|
return rows
|
|
f = dosen_filter.lower()
|
|
return [r for r in rows if f in r["nama"].lower() or f == r["scholar"].lower()]
|
|
|
|
|
|
def enrich_dosen(session, conn, dosen_rows, dry_run=False):
|
|
"""Isi dosen.sinta dan dosen.scholar yang masih kosong dari halaman
|
|
departemen SINTA (ID Scholar diambil dari URL foto profil)."""
|
|
if not any(not d["sinta"] or not d["scholar"] for d in dosen_rows):
|
|
return
|
|
for a in get_department_authors(session, DEFAULT_DEPT_URL):
|
|
d = match_dosen(a["nama"], dosen_rows)
|
|
if d is None:
|
|
continue
|
|
updates = {}
|
|
if not d["sinta"]:
|
|
d["sinta"] = updates["sinta"] = a["sinta_id"]
|
|
if not d["scholar"] and a["gs_id"]:
|
|
d["scholar"] = updates["scholar"] = a["gs_id"]
|
|
if updates and conn and d["id_dosen"] is not None and not dry_run:
|
|
with conn.cursor() as cur:
|
|
for col, val in updates.items():
|
|
cur.execute(f"UPDATE dosen SET {col}=%s WHERE id_dosen=%s", (val, d["id_dosen"]))
|
|
log.info("dosen '%s' dilengkapi: %s", d["nama"], updates)
|
|
if conn and not dry_run:
|
|
conn.commit()
|
|
|
|
|
|
def run_publications(dry_run=False, dosen_filter=None):
|
|
key = get_api_key()
|
|
budget = {"used": 0, "max": int(os.getenv("SERPAPI_MAX_REQUESTS", "90"))}
|
|
dosen_rows, db_ok = load_dosen()
|
|
conn = db_connect() if (not dry_run and db_ok) else None
|
|
if conn:
|
|
ensure_schema(conn)
|
|
enrich_dosen(make_session(), conn, dosen_rows, dry_run)
|
|
targets = _apply_filter([d for d in dosen_rows if d["scholar"]], dosen_filter)
|
|
log.info("%d dosen punya ID Scholar (dari %d)", len(targets), len(dosen_rows))
|
|
|
|
stats = {"authors": 0, "articles": 0, "inserted": 0, "skipped": 0, "errors": 0}
|
|
try:
|
|
existing = set()
|
|
if conn:
|
|
with conn.cursor() as cur:
|
|
cur.execute("SELECT id_author, judul FROM penelitian")
|
|
existing = {(r[0], norm_title(r[1] or "")) for r in cur.fetchall()}
|
|
|
|
for d in targets:
|
|
try:
|
|
gs_name, articles = fetch_author_articles(key, d["scholar"], budget)
|
|
except Exception as e:
|
|
log.error("Gagal tarik %s (%s): %s", d["nama"], d["scholar"], e)
|
|
stats["errors"] += 1
|
|
continue
|
|
stats["authors"] += 1
|
|
stats["articles"] += len(articles)
|
|
author_name = gs_name or d["nama"]
|
|
log.info("%s: %d artikel (request terpakai: %d)", author_name, len(articles), budget["used"])
|
|
if not conn:
|
|
continue
|
|
with conn.cursor() as cur:
|
|
for art in articles:
|
|
judul = (art.get("title") or "").strip()
|
|
if not judul:
|
|
continue
|
|
key_row = (d["scholar"], norm_title(judul))
|
|
if key_row in existing:
|
|
stats["skipped"] += 1
|
|
continue
|
|
ym = re.search(r"\d{4}", str(art.get("year") or ""))
|
|
cur.execute(
|
|
"INSERT INTO penelitian "
|
|
"(judul, author, link, tahun, author_full, id_author, biaya) "
|
|
"VALUES (%s, %s, %s, %s, %s, %s, '')",
|
|
(
|
|
judul[:400],
|
|
author_name[:100],
|
|
(art.get("citation_id") or "")[:300],
|
|
ym.group(0) if ym else "",
|
|
(art.get("authors") or "")[:255],
|
|
d["scholar"],
|
|
),
|
|
)
|
|
existing.add(key_row)
|
|
stats["inserted"] += 1
|
|
if conn:
|
|
conn.commit()
|
|
except Exception:
|
|
if conn:
|
|
conn.rollback()
|
|
raise
|
|
finally:
|
|
if conn:
|
|
conn.close()
|
|
|
|
stats["serpapi_requests"] = budget["used"]
|
|
log.info("Publications selesai: %s", stats)
|
|
return stats
|
|
|
|
|
|
# --------------------------------------------- langkah 2: kuartil dari SINTA
|
|
|
|
def run_scopus(dry_run=False, dosen_filter=None):
|
|
session = make_session()
|
|
dosen_rows, db_ok = load_dosen()
|
|
conn = db_connect() if db_ok else None
|
|
if conn:
|
|
ensure_schema(conn)
|
|
|
|
enrich_dosen(session, conn, dosen_rows, dry_run)
|
|
|
|
targets = _apply_filter([d for d in dosen_rows if d["sinta"]], dosen_filter)
|
|
log.info("%d dosen punya SINTA ID", len(targets))
|
|
|
|
stats = {"authors": 0, "scopus_items": 0, "updated": 0, "unmatched": 0}
|
|
pubs_by_author = {}
|
|
if conn:
|
|
with conn.cursor() as cur:
|
|
cur.execute("SELECT id_artikel, id_author, judul, scopus FROM penelitian")
|
|
for r in cur.fetchall():
|
|
pubs_by_author.setdefault(r[1], []).append(
|
|
{"id_artikel": r[0], "nt": norm_title(r[2] or ""), "scopus": r[3]}
|
|
)
|
|
try:
|
|
for d in targets:
|
|
items = get_author_publications(session, d["sinta"], view="scopus")
|
|
stats["authors"] += 1
|
|
stats["scopus_items"] += len(items)
|
|
if not conn:
|
|
continue
|
|
rows = pubs_by_author.get(d["scholar"], [])
|
|
with conn.cursor() as cur:
|
|
for it in items:
|
|
q = it["quartile"] or "Scopus"
|
|
match = next((r for r in rows if r["nt"] == norm_title(it["judul"])), None)
|
|
if not match:
|
|
stats["unmatched"] += 1
|
|
continue
|
|
if (match["scopus"] or "") != q and not dry_run:
|
|
cur.execute(
|
|
"UPDATE penelitian SET scopus=%s WHERE id_artikel=%s",
|
|
(q, match["id_artikel"]),
|
|
)
|
|
match["scopus"] = q
|
|
stats["updated"] += 1
|
|
if conn and not dry_run:
|
|
conn.commit()
|
|
except Exception:
|
|
if conn:
|
|
conn.rollback()
|
|
raise
|
|
finally:
|
|
if conn:
|
|
conn.close()
|
|
|
|
log.info("Scopus selesai: %s", stats)
|
|
return stats
|
|
|
|
|
|
def run(dry_run=False):
|
|
"""Entry point gabungan - dipakai CLI; Airflow memanggil per-langkah."""
|
|
return {
|
|
"publications": run_publications(dry_run=dry_run),
|
|
"scopus": run_scopus(dry_run=dry_run),
|
|
}
|
|
|
|
|
|
if __name__ == "__main__":
|
|
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
|
|
parser = argparse.ArgumentParser(description="Tarik penelitian via SerpAPI + kuartil SINTA")
|
|
parser.add_argument("--dry-run", action="store_true")
|
|
parser.add_argument("--only", choices=["publications", "scopus"])
|
|
parser.add_argument("--dosen", help="filter satu dosen: potongan nama atau ID Scholar")
|
|
args = parser.parse_args()
|
|
if args.only == "publications":
|
|
print(run_publications(dry_run=args.dry_run, dosen_filter=args.dosen))
|
|
elif args.only == "scopus":
|
|
print(run_scopus(dry_run=args.dry_run, dosen_filter=args.dosen))
|
|
else:
|
|
print(run_publications(dry_run=args.dry_run, dosen_filter=args.dosen))
|
|
print(run_scopus(dry_run=args.dry_run, dosen_filter=args.dosen))
|