Spaces:

taesiri
/

ClaudeReadsArxiv

Paused

App Files Files Community

taesiri commited on Oct 27, 2023

Commit

16c9cff

1 Parent(s): 2adf285

update

Browse files

Files changed (3) hide show

app.py +47 -61
packages.txt +8 -0
requirements.txt +2 -1

app.py CHANGED Viewed

@@ -8,6 +8,10 @@ import gradio as gr
 import requests
 import arxiv
 def replace_texttt(text):
@@ -28,58 +32,14 @@ def get_paper_info(paper_id):
         return None, None
-def download_arxiv_source(paper_id):
-    url = f"https://arxiv.org/e-print/{paper_id}"
-    # Get the tar file
-    response = requests.get(url)
-    response.raise_for_status()
-    # Open the tar file
-    tar = tarfile.open(fileobj=io.BytesIO(response.content), mode="r")
-    # Load all .tex files into memory, including their subdirectories
-    tex_files = {
-        member.name: tar.extractfile(member).read().decode("utf-8")
-        for member in tar.getmembers()
-        if member.name.endswith(".tex")
-    }
-    # Load all .tex files into memory, including their subdirectories
-    tex_files = {
-        member.name: tar.extractfile(member).read().decode("utf-8")
-        for member in tar.getmembers()
-        if member.isfile() and member.name.endswith(".tex")
-    }
-    # Pattern to match \input{filename} and \include{filename}
-    pattern = re.compile(r"\\(input|include){(.*?)}")
-    # Function to replace \input{filename} and \include{filename} with file contents
-    def replace_includes(text):
-        output = []
-        for line in text.split("\n"):
-            match = re.search(pattern, line)
-            if match:
-                command, filename = match.groups()
-                # LaTeX automatically adds .tex extension for \input and \include commands
-                if not filename.endswith(".tex"):
-                    filename += ".tex"
-                if filename in tex_files:
-                    output.append(replace_includes(tex_files[filename]))
-                else:
-                    output.append(f"% {line} % FILE NOT FOUND")
-            else:
-                output.append(line)
-        return "\n".join(output)
-    if "main.tex" in tex_files:
-        # Start with the contents of main.tex
-        main_tex = replace_includes(tex_files["main.tex"])
-    else:
-        # No main.tex, concatenate all .tex files
-        main_tex = "\n".join(replace_includes(text) for text in tex_files.values())
-    return main_tex
 class ContextualQA:
@@ -94,10 +54,28 @@ class ContextualQA:
         self.context = text
     def ask_question(self, question):
-        leading_prompt = "Give the following paper:"
-        trailing_prompt = "Now, answer the following question based on the content of the paper above. You can optionally use Markdown to format your answer or LaTeX typesetting to improve the presentation of your answer."
-        prompt = f"{HUMAN_PROMPT} {leading_prompt} {self.context} {trailing_prompt} {HUMAN_PROMPT} {question} {AI_PROMPT}"
         response = self.client.completions.create(
             prompt=prompt,
             stop_sequences=[HUMAN_PROMPT],
@@ -126,14 +104,22 @@ class ContextualQA:
 def load_context(paper_id):
-    try:
-        latex_source = download_arxiv_source(paper_id)
-    except Exception as e:
-        return None, [(f"Error loading paper with id {paper_id}.", str(e))]
     client = Anthropic(api_key=os.environ["ANTHROPIC_API_KEY"])
     qa_model = ContextualQA(client, model="claude-2.0")
-    qa_model.load_text(latex_source)
     # Usage
     title, abstract = get_paper_info(paper_id)
@@ -146,7 +132,7 @@ def load_context(paper_id):
         [
             (
                 f"Load the paper with id {paper_id}.",
-                f"\n**Title**: {title}\n\n**Abstract**: {abstract}\n\nPaper loaded, You can now ask questions.",
             )
         ],
     )

 import requests
 import arxiv
+from arxiv_latex_extractor import get_paper_content
+import requests
+LEADING_PROMPT = "Read the following paper and answer the question below:"
 def replace_texttt(text):
         return None, None
+def get_paper_from_huggingface(paper_id):
+    try:
+        url = f"https://huggingface.co/datasets/taesiri/arxiv_db/raw/main/papers/{paper_id}.tex"
+        response = requests.get(url)
+        response.raise_for_status()  # Will raise an HTTPError if the HTTP request returned an unsuccessful status code
+        return response.text
+    except Exception as e:
+        return None
 class ContextualQA:
         self.context = text
     def ask_question(self, question):
+        if self.questions:
+            # For the first question-answer pair, don't add HUMAN_PROMPT before the question
+            first_pair = f"Question: {self.questions[0]}\n{AI_PROMPT} Answer: {self.responses[0]}"
+            # For subsequent questions, include both HUMAN_PROMPT and AI_PROMPT
+            subsequent_pairs = "\n".join(
+                [
+                    f"{HUMAN_PROMPT} Question: {q}\n{AI_PROMPT} Answer: {a}"
+                    for q, a in zip(self.questions[1:], self.responses[1:])
+                ]
+            )
+            history_context = f"{first_pair}\n{subsequent_pairs}"
+        else:
+            history_context = ""
+        full_context = f"{self.context}\n\n{history_context}\n"
+        prompt = f"{HUMAN_PROMPT}  {full_context} {HUMAN_PROMPT} {question} {AI_PROMPT}"
+        # save prompt on disk for examination
+        with open("prompt.txt", "w") as f:
+            f.write(prompt)
         response = self.client.completions.create(
             prompt=prompt,
             stop_sequences=[HUMAN_PROMPT],
 def load_context(paper_id):
+    global LEADING_PROMPT
+    # First, try to get the paper from Hugging Face
+    latex_source = get_paper_from_huggingface(paper_id)
+    # If not found, use arxiv_latex_extractor
+    if not latex_source:
+        try:
+            latex_source = get_paper_content(paper_id)
+        except Exception as e:
+            return None, [(f"Error loading paper with id {paper_id}: {e}",)]
     client = Anthropic(api_key=os.environ["ANTHROPIC_API_KEY"])
     qa_model = ContextualQA(client, model="claude-2.0")
+    context = f"{LEADING_PROMPT}\n{latex_source}"
+    qa_model.load_text(context)
     # Usage
     title, abstract = get_paper_info(paper_id)
         [
             (
                 f"Load the paper with id {paper_id}.",
+                f"\n**Title**: {title}\n\n**Abstract**: {abstract}\n\nPaper loaded. You can now ask questions.",
             )
         ],
     )

packages.txt ADDED Viewed

	@@ -0,0 +1,8 @@

+perl
+cpanminus
+texlive-full
+texlive-fonts-extra
+texlive-font-utils
+pandoc
+poppler-utils
+unzip

requirements.txt CHANGED Viewed

@@ -6,4 +6,5 @@ seaborn
 tqdm
 numpy
 arxiv
-tiktoken

 tqdm
 numpy
 arxiv
+tiktoken
+git+https://github.com/taesiri/arxiv_latex_extractor