research-agent / app.py
APPLE
Initial commit: Research Agent
5ae917c
Raw History Blame Contribute Delete
8.68 kB
from flask import Flask, request, jsonify
from flask_cors import CORS
import requests
import re
import xml.etree.ElementTree as ET
app = Flask(__name__, static_folder=".", static_url_path="")
CORS(app)
def search_semantic_scholar(query, limit=8):
try:
url = "https://api.semanticscholar.org/graph/v1/paper/search"
params = {
"query": query,
"limit": limit,
"fields": "title,authors,year,abstract,citationCount,externalIds,venue,openAccessPdf,url"
}
r = requests.get(url, params=params, timeout=10)
r.raise_for_status()
papers = []
for p in r.json().get("data", []):
authors = [a.get("name", "") for a in (p.get("authors") or [])[:4]]
doi = (p.get("externalIds") or {}).get("DOI")
url_out = f"https://doi.org/{doi}" if doi else p.get("url", "")
papers.append({
"title": p.get("title", ""),
"authors": authors,
"year": p.get("year"),
"abstract": (p.get("abstract") or "")[:400],
"url": url_out,
"source": "Semantic Scholar",
"venue": p.get("venue", ""),
"citations": p.get("citationCount"),
})
return papers
except:
return []
def search_openalex(query, limit=8):
try:
url = "https://api.openalex.org/works"
params = {
"search": query,
"per-page": limit,
"select": "title,authorships,publication_year,abstract_inverted_index,cited_by_count,primary_location,doi,id",
"mailto": "research-agent@app.com",
}
r = requests.get(url, params=params, timeout=10)
r.raise_for_status()
papers = []
for p in r.json().get("results", []):
abstract = ""
inv = p.get("abstract_inverted_index") or {}
if inv:
try:
words = [""] * (max(max(v) for v in inv.values()) + 1)
for word, positions in inv.items():
for pos in positions:
words[pos] = word
abstract = " ".join(words)[:400]
except:
pass
authors = []
for a in (p.get("authorships") or [])[:4]:
name = (a.get("author") or {}).get("display_name", "")
if name:
authors.append(name)
doi = p.get("doi", "")
venue = ((p.get("primary_location") or {}).get("source") or {}).get("display_name", "")
url_out = f"https://doi.org/{doi.replace('https://doi.org/','')}" if doi else p.get("id", "")
papers.append({
"title": p.get("title", ""),
"authors": authors,
"year": p.get("publication_year"),
"abstract": abstract,
"url": url_out,
"source": "OpenAlex",
"venue": venue,
"citations": p.get("cited_by_count"),
})
return papers
except:
return []
def search_arxiv(query, limit=6):
try:
url = "http://export.arxiv.org/api/query"
params = {"search_query": f"all:{query}", "max_results": limit, "sortBy": "relevance"}
r = requests.get(url, params=params, timeout=10)
r.raise_for_status()
ns = "{http://www.w3.org/2005/Atom}"
root = ET.fromstring(r.text)
papers = []
for entry in root.findall(f"{ns}entry"):
title = re.sub(r"\s+", " ", (entry.find(f"{ns}title") or type("x", (), {"text": ""})()).text or "").strip()
abstract = re.sub(r"\s+", " ", (entry.find(f"{ns}summary") or type("x", (), {"text": ""})()).text or "").strip()[:400]
link = (entry.find(f"{ns}id") or type("x", (), {"text": ""})()).text or ""
year = None
published = entry.find(f"{ns}published")
if published is not None and published.text:
year = int(published.text[:4])
authors = []
for author in entry.findall(f"{ns}author")[:4]:
name = (author.find(f"{ns}name") or type("x", (), {"text": None})()).text
if name:
authors.append(name)
papers.append({
"title": title,
"authors": authors,
"year": year,
"abstract": abstract,
"url": link,
"source": "arXiv",
"venue": "arXiv preprint",
"citations": None,
})
return papers
except:
return []
def search_pubmed(query, limit=6):
try:
r = requests.get("https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esearch.fcgi",
params={"db": "pubmed", "term": query, "retmax": limit, "retmode": "json"}, timeout=10)
r.raise_for_status()
ids = r.json().get("esearchresult", {}).get("idlist", [])
if not ids:
return []
r2 = requests.get("https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esummary.fcgi",
params={"db": "pubmed", "id": ",".join(ids), "retmode": "json"}, timeout=10)
r2.raise_for_status()
result = r2.json().get("result", {})
papers = []
for pid in ids:
p = result.get(pid, {})
year = None
m = re.search(r"\d{4}", p.get("pubdate", ""))
if m:
year = int(m.group())
papers.append({
"title": p.get("title", ""),
"authors": [a.get("name", "") for a in (p.get("authors") or [])[:4]],
"year": year,
"abstract": "",
"url": f"https://pubmed.ncbi.nlm.nih.gov/{pid}/",
"source": "PubMed",
"venue": p.get("source", ""),
"citations": None,
})
return papers
except:
return []
def search_ieee(query, limit=6):
"""IEEE Xplore open metadata search (no API key needed for basic queries)."""
try:
url = "https://ieeexploreapi.ieee.org/api/v1/search/articles"
params = {
"querytext": query,
"max_records": limit,
"start_record": 1,
"sort_order": "desc",
"sort_field": "article_number",
"apikey": "jdst3gm3nkyb2e4zj6v7nq5a", # public demo key
}
r = requests.get(url, params=params, timeout=10)
r.raise_for_status()
papers = []
for p in r.json().get("articles", []):
authors = [a.get("full_name", "") for a in (p.get("authors", {}).get("authors") or [])[:4]]
year = None
try:
year = int(p.get("publication_year", ""))
except:
pass
papers.append({
"title": p.get("title", ""),
"authors": authors,
"year": year,
"abstract": (p.get("abstract") or "")[:400],
"url": p.get("html_url", "") or p.get("pdf_url", ""),
"source": "IEEE Xplore",
"venue": p.get("publication_title", ""),
"citations": p.get("citing_paper_count"),
})
return papers
except:
return []
def deduplicate(papers):
seen, out = set(), []
for p in papers:
key = re.sub(r"[^a-z0-9]", "", (p.get("title") or "").lower())[:60]
if key and key not in seen:
seen.add(key)
out.append(p)
return out
@app.route("/search", methods=["POST"])
def search():
data = request.json
query = data.get("query", "")
sources = data.get("sources", ["Semantic Scholar", "OpenAlex", "arXiv", "PubMed", "IEEE Xplore"])
limit = data.get("limit", 8)
all_papers = []
for source in sources:
if source == "Semantic Scholar":
all_papers.extend(search_semantic_scholar(query, limit))
elif source == "OpenAlex":
all_papers.extend(search_openalex(query, limit))
elif source == "arXiv":
all_papers.extend(search_arxiv(query, min(limit, 6)))
elif source == "PubMed":
all_papers.extend(search_pubmed(query, min(limit, 6)))
elif source == "IEEE Xplore":
all_papers.extend(search_ieee(query, min(limit, 6)))
return jsonify(deduplicate(all_papers))
@app.route("/")
def index():
return app.send_static_file("index.html")
if __name__ == "__main__":
app.run(host="0.0.0.0", port=7860, debug=False)