summaryrefslogtreecommitdiff
path: root/scripts/extract_abstract
diff options
context:
space:
mode:
authorCaptainJack2491 <jayrupnakawala@gmail.com>2026-01-09 18:37:34 +0000
committerCaptainJack2491 <jayrupnakawala@gmail.com>2026-01-09 18:37:34 +0000
commit40da5b48beeda6418c802c65a0dc8bf85cc9aca4 (patch)
tree4f43306abba9910aad98435580cb6506dae4a698 /scripts/extract_abstract
parent46ee0a5914d8248976c7e2308ce16f32f1bd4783 (diff)
added scripts to get abstract and bibliography easily from papers
Diffstat (limited to 'scripts/extract_abstract')
-rwxr-xr-xscripts/extract_abstract42
1 files changed, 42 insertions, 0 deletions
diff --git a/scripts/extract_abstract b/scripts/extract_abstract
new file mode 100755
index 0000000..e70b3cb
--- /dev/null
+++ b/scripts/extract_abstract
@@ -0,0 +1,42 @@
+#!/bin/bash
+
+# Directory containing the papers
+PAPER_DIR="../papers"
+
+echo "Fetching abstracts for arXiv papers in $PAPER_DIR..."
+echo "===================================================="
+
+for file in "$PAPER_DIR"/*.pdf; do
+ # Extract the arXiv ID (e.g., 2401.05566) from the filename
+ filename=$(basename "$file")
+ id=$(echo "$filename" | grep -oE '[0-9]{4}\.[0-9]{4,5}')
+
+ if [ -n "$id" ]; then
+ echo "FILE: $filename"
+ echo "ID: $id"
+
+ # Fetch XML from arXiv API and parse the <summary> tag using Python
+ curl -s "https://export.arxiv.org/api/query?id_list=$id" | \
+ python3 -c "
+import sys, xml.etree.ElementTree as ET
+try:
+ xml_data = sys.stdin.read()
+ root = ET.fromstring(xml_data)
+ ns = {'atom': 'http://www.w3.org/2005/Atom'}
+ summary = root.find('.//atom:summary', ns)
+ if summary is not None:
+ print(summary.text.strip())
+ else:
+ print('Abstract not found in API response.')
+except Exception as e:
+ print(f'Error parsing response: {e}')
+"
+ echo "----------------------------------------------------"
+ else
+ echo "Skipping: $filename (No arXiv ID detected)"
+ echo "----------------------------------------------------"
+ fi
+ sleep 1
+done
+
+