diff options
Diffstat (limited to 'scripts')
| -rwxr-xr-x | scripts/extract_abstract | 42 | ||||
| -rwxr-xr-x | scripts/generate_refrences | 30 |
2 files changed, 72 insertions, 0 deletions
diff --git a/scripts/extract_abstract b/scripts/extract_abstract new file mode 100755 index 0000000..e70b3cb --- /dev/null +++ b/scripts/extract_abstract @@ -0,0 +1,42 @@ +#!/bin/bash + +# Directory containing the papers +PAPER_DIR="../papers" + +echo "Fetching abstracts for arXiv papers in $PAPER_DIR..." +echo "====================================================" + +for file in "$PAPER_DIR"/*.pdf; do + # Extract the arXiv ID (e.g., 2401.05566) from the filename + filename=$(basename "$file") + id=$(echo "$filename" | grep -oE '[0-9]{4}\.[0-9]{4,5}') + + if [ -n "$id" ]; then + echo "FILE: $filename" + echo "ID: $id" + + # Fetch XML from arXiv API and parse the <summary> tag using Python + curl -s "https://export.arxiv.org/api/query?id_list=$id" | \ + python3 -c " +import sys, xml.etree.ElementTree as ET +try: + xml_data = sys.stdin.read() + root = ET.fromstring(xml_data) + ns = {'atom': 'http://www.w3.org/2005/Atom'} + summary = root.find('.//atom:summary', ns) + if summary is not None: + print(summary.text.strip()) + else: + print('Abstract not found in API response.') +except Exception as e: + print(f'Error parsing response: {e}') +" + echo "----------------------------------------------------" + else + echo "Skipping: $filename (No arXiv ID detected)" + echo "----------------------------------------------------" + fi + sleep 1 +done + + diff --git a/scripts/generate_refrences b/scripts/generate_refrences new file mode 100755 index 0000000..1d5ec25 --- /dev/null +++ b/scripts/generate_refrences @@ -0,0 +1,30 @@ +#!/bin/bash + +OUTPUT_FILE="../src/final/references.bib" +echo "Generating $OUTPUT_FILE..." + +# Clear the file if it exists +> "$OUTPUT_FILE" + +for file in ../papers/*.pdf; do + # Extract the arXiv ID (e.g., 2401.05566) + id=$(basename "$file" | grep -oE '[0-9]{4}\.[0-9]{4,5}') + + if [ -n "$id" ]; then + echo "Fetching BibTeX for ID: $id" + + # Fetch the BibTeX entry from arXiv's export service + # -L follows redirects + # -s is silent + curl -L -s "https://arxiv.org/bibtex/$id" >> "$OUTPUT_FILE" + + # Add a newline between entries + echo -e "\n" >> "$OUTPUT_FILE" + else + echo "Skipping non-arXiv file: $(basename "$file")" + fi +done + +echo "Done! Created $OUTPUT_FILE with $(grep -c "@article" "$OUTPUT_FILE") entries." + + |
