Spaces:
Running
Running
Download app.py from elprofessor15/research-agent: direct link, hf CLI and curl.
- Browser
- Download file 8.68 kB
-
https://huggingface.co/spaces/elprofessor15/research-agent/resolve/main/app.py
- Command line
-
hf download hf://spaces/elprofessor15/research-agent/app.py
-
curl -L -o app.py https://huggingface.co/spaces/elprofessor15/research-agent/resolve/main/app.py
8.68 kB
| from flask import Flask, request, jsonify | |
| from flask_cors import CORS | |
| import requests | |
| import re | |
| import xml.etree.ElementTree as ET | |
| app = Flask(__name__, static_folder=".", static_url_path="") | |
| CORS(app) | |
| def search_semantic_scholar(query, limit=8): | |
| try: | |
| url = "https://api.semanticscholar.org/graph/v1/paper/search" | |
| params = { | |
| "query": query, | |
| "limit": limit, | |
| "fields": "title,authors,year,abstract,citationCount,externalIds,venue,openAccessPdf,url" | |
| } | |
| r = requests.get(url, params=params, timeout=10) | |
| r.raise_for_status() | |
| papers = [] | |
| for p in r.json().get("data", []): | |
| authors = [a.get("name", "") for a in (p.get("authors") or [])[:4]] | |
| doi = (p.get("externalIds") or {}).get("DOI") | |
| url_out = f"https://doi.org/{doi}" if doi else p.get("url", "") | |
| papers.append({ | |
| "title": p.get("title", ""), | |
| "authors": authors, | |
| "year": p.get("year"), | |
| "abstract": (p.get("abstract") or "")[:400], | |
| "url": url_out, | |
| "source": "Semantic Scholar", | |
| "venue": p.get("venue", ""), | |
| "citations": p.get("citationCount"), | |
| }) | |
| return papers | |
| except: | |
| return [] | |
| def search_openalex(query, limit=8): | |
| try: | |
| url = "https://api.openalex.org/works" | |
| params = { | |
| "search": query, | |
| "per-page": limit, | |
| "select": "title,authorships,publication_year,abstract_inverted_index,cited_by_count,primary_location,doi,id", | |
| "mailto": "research-agent@app.com", | |
| } | |
| r = requests.get(url, params=params, timeout=10) | |
| r.raise_for_status() | |
| papers = [] | |
| for p in r.json().get("results", []): | |
| abstract = "" | |
| inv = p.get("abstract_inverted_index") or {} | |
| if inv: | |
| try: | |
| words = [""] * (max(max(v) for v in inv.values()) + 1) | |
| for word, positions in inv.items(): | |
| for pos in positions: | |
| words[pos] = word | |
| abstract = " ".join(words)[:400] | |
| except: | |
| pass | |
| authors = [] | |
| for a in (p.get("authorships") or [])[:4]: | |
| name = (a.get("author") or {}).get("display_name", "") | |
| if name: | |
| authors.append(name) | |
| doi = p.get("doi", "") | |
| venue = ((p.get("primary_location") or {}).get("source") or {}).get("display_name", "") | |
| url_out = f"https://doi.org/{doi.replace('https://doi.org/','')}" if doi else p.get("id", "") | |
| papers.append({ | |
| "title": p.get("title", ""), | |
| "authors": authors, | |
| "year": p.get("publication_year"), | |
| "abstract": abstract, | |
| "url": url_out, | |
| "source": "OpenAlex", | |
| "venue": venue, | |
| "citations": p.get("cited_by_count"), | |
| }) | |
| return papers | |
| except: | |
| return [] | |
| def search_arxiv(query, limit=6): | |
| try: | |
| url = "http://export.arxiv.org/api/query" | |
| params = {"search_query": f"all:{query}", "max_results": limit, "sortBy": "relevance"} | |
| r = requests.get(url, params=params, timeout=10) | |
| r.raise_for_status() | |
| ns = "{http://www.w3.org/2005/Atom}" | |
| root = ET.fromstring(r.text) | |
| papers = [] | |
| for entry in root.findall(f"{ns}entry"): | |
| title = re.sub(r"\s+", " ", (entry.find(f"{ns}title") or type("x", (), {"text": ""})()).text or "").strip() | |
| abstract = re.sub(r"\s+", " ", (entry.find(f"{ns}summary") or type("x", (), {"text": ""})()).text or "").strip()[:400] | |
| link = (entry.find(f"{ns}id") or type("x", (), {"text": ""})()).text or "" | |
| year = None | |
| published = entry.find(f"{ns}published") | |
| if published is not None and published.text: | |
| year = int(published.text[:4]) | |
| authors = [] | |
| for author in entry.findall(f"{ns}author")[:4]: | |
| name = (author.find(f"{ns}name") or type("x", (), {"text": None})()).text | |
| if name: | |
| authors.append(name) | |
| papers.append({ | |
| "title": title, | |
| "authors": authors, | |
| "year": year, | |
| "abstract": abstract, | |
| "url": link, | |
| "source": "arXiv", | |
| "venue": "arXiv preprint", | |
| "citations": None, | |
| }) | |
| return papers | |
| except: | |
| return [] | |
| def search_pubmed(query, limit=6): | |
| try: | |
| r = requests.get("https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esearch.fcgi", | |
| params={"db": "pubmed", "term": query, "retmax": limit, "retmode": "json"}, timeout=10) | |
| r.raise_for_status() | |
| ids = r.json().get("esearchresult", {}).get("idlist", []) | |
| if not ids: | |
| return [] | |
| r2 = requests.get("https://eutils.ncbi.nlm.nih.gov/entrez/eutils/esummary.fcgi", | |
| params={"db": "pubmed", "id": ",".join(ids), "retmode": "json"}, timeout=10) | |
| r2.raise_for_status() | |
| result = r2.json().get("result", {}) | |
| papers = [] | |
| for pid in ids: | |
| p = result.get(pid, {}) | |
| year = None | |
| m = re.search(r"\d{4}", p.get("pubdate", "")) | |
| if m: | |
| year = int(m.group()) | |
| papers.append({ | |
| "title": p.get("title", ""), | |
| "authors": [a.get("name", "") for a in (p.get("authors") or [])[:4]], | |
| "year": year, | |
| "abstract": "", | |
| "url": f"https://pubmed.ncbi.nlm.nih.gov/{pid}/", | |
| "source": "PubMed", | |
| "venue": p.get("source", ""), | |
| "citations": None, | |
| }) | |
| return papers | |
| except: | |
| return [] | |
| def search_ieee(query, limit=6): | |
| """IEEE Xplore open metadata search (no API key needed for basic queries).""" | |
| try: | |
| url = "https://ieeexploreapi.ieee.org/api/v1/search/articles" | |
| params = { | |
| "querytext": query, | |
| "max_records": limit, | |
| "start_record": 1, | |
| "sort_order": "desc", | |
| "sort_field": "article_number", | |
| "apikey": "jdst3gm3nkyb2e4zj6v7nq5a", # public demo key | |
| } | |
| r = requests.get(url, params=params, timeout=10) | |
| r.raise_for_status() | |
| papers = [] | |
| for p in r.json().get("articles", []): | |
| authors = [a.get("full_name", "") for a in (p.get("authors", {}).get("authors") or [])[:4]] | |
| year = None | |
| try: | |
| year = int(p.get("publication_year", "")) | |
| except: | |
| pass | |
| papers.append({ | |
| "title": p.get("title", ""), | |
| "authors": authors, | |
| "year": year, | |
| "abstract": (p.get("abstract") or "")[:400], | |
| "url": p.get("html_url", "") or p.get("pdf_url", ""), | |
| "source": "IEEE Xplore", | |
| "venue": p.get("publication_title", ""), | |
| "citations": p.get("citing_paper_count"), | |
| }) | |
| return papers | |
| except: | |
| return [] | |
| def deduplicate(papers): | |
| seen, out = set(), [] | |
| for p in papers: | |
| key = re.sub(r"[^a-z0-9]", "", (p.get("title") or "").lower())[:60] | |
| if key and key not in seen: | |
| seen.add(key) | |
| out.append(p) | |
| return out | |
| def search(): | |
| data = request.json | |
| query = data.get("query", "") | |
| sources = data.get("sources", ["Semantic Scholar", "OpenAlex", "arXiv", "PubMed", "IEEE Xplore"]) | |
| limit = data.get("limit", 8) | |
| all_papers = [] | |
| for source in sources: | |
| if source == "Semantic Scholar": | |
| all_papers.extend(search_semantic_scholar(query, limit)) | |
| elif source == "OpenAlex": | |
| all_papers.extend(search_openalex(query, limit)) | |
| elif source == "arXiv": | |
| all_papers.extend(search_arxiv(query, min(limit, 6))) | |
| elif source == "PubMed": | |
| all_papers.extend(search_pubmed(query, min(limit, 6))) | |
| elif source == "IEEE Xplore": | |
| all_papers.extend(search_ieee(query, min(limit, 6))) | |
| return jsonify(deduplicate(all_papers)) | |
| def index(): | |
| return app.send_static_file("index.html") | |
| if __name__ == "__main__": | |
| app.run(host="0.0.0.0", port=7860, debug=False) |