from flask import Flask, request, jsonify from flask_cors import CORS import requests import re import xml.etree.ElementTree as ET app = Flask(__name__, static_folder=".", static_url_path="") CORS(app) def search_semantic_scholar(query, limit=8): try: url = "https://api.semanticscholar.org/graph/v1/paper/search" params = { "query": query, "limit": limit, "fields": "title,authors,year,abstract,citationCount,externalIds,venue,openAccessPdf,url" } r = requests.get(url, params=params, timeout=10) r.raise_for_status() papers = [] for p in r.json().get("data", []): authors = [a.get("name", "") for a in (p.get("authors") or [])[:4]] doi = (p.get("externalIds") or {}).get("DOI") url_out = f"https://doi.org/{doi}" if doi else p.get("url", "") papers.append({ "title": p.get("title", ""), "authors": authors, "year": p.get("year"), "abstract": (p.get("abstract") or "")[:400], "url": url_out, "source": "Semantic Scholar", "venue": p.get("venue", ""), "citations": p.get("citationCount"), }) return papers except: return [] def search_openalex(query, limit=8): try: url = "https://api.openalex.org/works" params = { "search": query, "per-page": limit, "select": "title,authorships,publication_year,abstract_inverted_index,cited_by_count,primary_location,doi,id", "mailto": "research-agent@app.com", } r = requests.get(url, params=params, timeout=10) r.raise_for_status() papers = [] for p in r.json().get("results", []): abstract = "" inv = p.get("abstract_inverted_index") or {} if inv: try: words = [""] * (max(max(v) for v in inv.values()) + 1) for word, positions in inv.items(): for pos in positions: words[pos] = word abstract = " ".join(words)[:400] except: pass authors = [] for a in (p.get("authorships") or [])[:4]: name = (a.get("author") or {}).get("display_name", "") if name: authors.append(name) doi = p.get("doi", "") venue = ((p.get("primary_location") or {}).get("source") or {}).get("display_name", "") url_out = f"https://doi.org/{doi.replace('https://doi.org/','')}" if doi else p.get("id", "") papers.append({ "title": p.get("title", ""), "authors": authors, "year": p.get("publication_year"), "abstract": abstract, "url": url_out, "source": "OpenAlex", "venue": venue, "citations": p.get("cited_by_count"), }) return papers except: return [] def search_arxiv(query, limit=6): try: url = "http://export.arxiv.org/api/query" params = {"search_query": f"all:{query}", "max_results": limit, "sortBy": "relevance"} r = requests.get(url, params=params, timeout=10) r.raise_for_status() ns = "{http://www.w3.org/2005/Atom}" root = ET.fromstring(r.text) papers = [] for entry in root.findall(f"{ns}entry"): title = re.sub(r"\s+", " ", (entry.find(f"{ns}title") or type("x", (), {"text": ""})()).text or "").strip() abstract = re.sub(r"\s+", " ", (entry.find(f"{ns}summary") or type("x", (), {"text": ""})()).text or "").strip()[:400] link = (entry.find(f"{ns}id") or type("x", (), {"text": ""})()).text or "" year = None published = entry.find(f"{ns}published") if published is not None and published.text: year = int(published.text[:4]) authors = [] for author in entry.findall(f"{ns}author")[:4]: name = (author.find(f"{ns}name") or type("x", (), {"text": None})()).text if name: authors.append(name) papers.append({ "title": title, "authors": authors, "year": year, "abstract": abstract, "url": link, "source": "arXiv", "venue": "arXiv preprint", "citations": None, }) return papers except: return [] def search_pubmed(query, limit=6): try: r = requests.get("https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esearch.fcgi", params={"db": "pubmed", "term": query, "retmax": limit, "retmode": "json"}, timeout=10) r.raise_for_status() ids = r.json().get("esearchresult", {}).get("idlist", []) if not ids: return [] r2 = requests.get("https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esummary.fcgi", params={"db": "pubmed", "id": ",".join(ids), "retmode": "json"}, timeout=10) r2.raise_for_status() result = r2.json().get("result", {}) papers = [] for pid in ids: p = result.get(pid, {}) year = None m = re.search(r"\d{4}", p.get("pubdate", "")) if m: year = int(m.group()) papers.append({ "title": p.get("title", ""), "authors": [a.get("name", "") for a in (p.get("authors") or [])[:4]], "year": year, "abstract": "", "url": f"https://pubmed.ncbi.nlm.nih.gov/{pid}/", "source": "PubMed", "venue": p.get("source", ""), "citations": None, }) return papers except: return [] def search_ieee(query, limit=6): """IEEE Xplore open metadata search (no API key needed for basic queries).""" try: url = "https://ieeexploreapi.ieee.org/api/v1/search/articles" params = { "querytext": query, "max_records": limit, "start_record": 1, "sort_order": "desc", "sort_field": "article_number", "apikey": "jdst3gm3nkyb2e4zj6v7nq5a", # public demo key } r = requests.get(url, params=params, timeout=10) r.raise_for_status() papers = [] for p in r.json().get("articles", []): authors = [a.get("full_name", "") for a in (p.get("authors", {}).get("authors") or [])[:4]] year = None try: year = int(p.get("publication_year", "")) except: pass papers.append({ "title": p.get("title", ""), "authors": authors, "year": year, "abstract": (p.get("abstract") or "")[:400], "url": p.get("html_url", "") or p.get("pdf_url", ""), "source": "IEEE Xplore", "venue": p.get("publication_title", ""), "citations": p.get("citing_paper_count"), }) return papers except: return [] def deduplicate(papers): seen, out = set(), [] for p in papers: key = re.sub(r"[^a-z0-9]", "", (p.get("title") or "").lower())[:60] if key and key not in seen: seen.add(key) out.append(p) return out @app.route("/search", methods=["POST"]) def search(): data = request.json query = data.get("query", "") sources = data.get("sources", ["Semantic Scholar", "OpenAlex", "arXiv", "PubMed", "IEEE Xplore"]) limit = data.get("limit", 8) all_papers = [] for source in sources: if source == "Semantic Scholar": all_papers.extend(search_semantic_scholar(query, limit)) elif source == "OpenAlex": all_papers.extend(search_openalex(query, limit)) elif source == "arXiv": all_papers.extend(search_arxiv(query, min(limit, 6))) elif source == "PubMed": all_papers.extend(search_pubmed(query, min(limit, 6))) elif source == "IEEE Xplore": all_papers.extend(search_ieee(query, min(limit, 6))) return jsonify(deduplicate(all_papers)) @app.route("/") def index(): return app.send_static_file("index.html") if __name__ == "__main__": app.run(host="0.0.0.0", port=7860, debug=False)