# Document Comparison

This notebook shows how to use an agent to compare two documents.

The high level idea is we will create a question-answering chain for each document, and then use that 

In [1]:
from pydantic import BaseModel, Field

from langchain.chat_models import ChatOpenAI
from langchain.agents import Tool
from langchain.embeddings.openai import OpenAIEmbeddings
from langchain.text_splitter import CharacterTextSplitter
from langchain.vectorstores import FAISS
from langchain.document_loaders import PyPDFLoader
from langchain.chains import RetrievalQA

In [2]:
class DocumentInput(BaseModel):
    question: str = Field()


llm = ChatOpenAI(temperature=0, model="gpt-3.5-turbo-0613")

tools = []
files = [
    # https://abc.xyz/investor/static/pdf/2023Q1_alphabet_earnings_release.pdf
    {
        "name": "alphabet-earnings",
        "path": "/Users/harrisonchase/Downloads/2023Q1_alphabet_earnings_release.pdf",
    },
    # https://digitalassets.tesla.com/tesla-contents/image/upload/IR/TSLA-Q1-2023-Update
    {
        "name": "tesla-earnings",
        "path": "/Users/harrisonchase/Downloads/TSLA-Q1-2023-Update.pdf",
    },
]

for file in files:
    loader = PyPDFLoader(file["path"])
    pages = loader.load_and_split()
    text_splitter = CharacterTextSplitter(chunk_size=1000, chunk_overlap=0)
    docs = text_splitter.split_documents(pages)
    embeddings = OpenAIEmbeddings()
    retriever = FAISS.from_documents(docs, embeddings).as_retriever()

    # Wrap retrievers in a Tool
    tools.append(
        Tool(
            args_schema=DocumentInput,
            name=file["name"],
            description=f"useful when you want to answer questions about {file['name']}",
            func=RetrievalQA.from_chain_type(llm=llm, retriever=retriever),
        )
    )

In [3]:
from langchain.agents import initialize_agent
from langchain.agents import AgentType

In [4]:
llm = ChatOpenAI(
    temperature=0,
    model="gpt-3.5-turbo-0613",
)

agent = initialize_agent(
    agent=AgentType.OPENAI_FUNCTIONS,
    tools=tools,
    llm=llm,
    verbose=True,
)

agent({"input": "did alphabet or tesla have more revenue?"})



[1m> Entering new  chain...[0m
[32;1m[1;3m
Invoking: `alphabet-earnings` with `{'question': 'revenue'}`


[0m[36;1m[1;3m{'query': 'revenue', 'result': 'The revenue for Alphabet Inc. for the quarter ended March 31, 2023, was $69,787 million.'}[0m[32;1m[1;3m
Invoking: `tesla-earnings` with `{'question': 'revenue'}`


[0m[33;1m[1;3m{'query': 'revenue', 'result': 'Total revenue for Q1-2023 was $23.3 billion.'}[0m[32;1m[1;3mAlphabet Inc. had more revenue than Tesla. Alphabet's revenue for the quarter ended March 31, 2023, was $69,787 million, while Tesla's total revenue for Q1-2023 was $23.3 billion.[0m

[1m> Finished chain.[0m


{'input': 'did alphabet or tesla have more revenue?',
 'output': "Alphabet Inc. had more revenue than Tesla. Alphabet's revenue for the quarter ended March 31, 2023, was $69,787 million, while Tesla's total revenue for Q1-2023 was $23.3 billion."}

## OpenAI Multi Functions

This type of agent allows calling multiple functions at once. This is really useful when some steps can be computed in parallel - like when asked to compare multiple documents

In [5]:
import langchain

langchain.debug = True

In [6]:
llm = ChatOpenAI(
    temperature=0,
    model="gpt-3.5-turbo-0613",
)

agent = initialize_agent(
    agent=AgentType.OPENAI_MULTI_FUNCTIONS,
    tools=tools,
    llm=llm,
    verbose=True,
)

agent({"input": "did alphabet or tesla have more revenue?"})

[32;1m[1;3m[chain/start][0m [1m[1:chain:AgentExecutor] Entering Chain run with input:
[0m{
  "input": "did alphabet or tesla have more revenue?"
}
[32;1m[1;3m[llm/start][0m [1m[1:chain:AgentExecutor > 2:llm:ChatOpenAI] Entering LLM run with input:
[0m{
  "prompts": [
    "System: You are a helpful AI assistant.\nHuman: did alphabet or tesla have more revenue?"
  ]
}
[36;1m[1;3m[llm/end][0m [1m[1:chain:AgentExecutor > 2:llm:ChatOpenAI] [2.66s] Exiting LLM run with output:
[0m{
  "generations": [
    [
      {
        "text": "",
        "generation_info": null,
        "message": {
          "content": "",
          "additional_kwargs": {
            "function_call": {
              "name": "tool_selection",
              "arguments": "{\n  \"actions\": [\n    {\n      \"action_name\": \"alphabet-earnings\",\n      \"action\": {\n        \"question\": \"What was Alphabet's revenue?\"\n      }\n    },\n    {\n      \"action_name\": \"tesla-earnings\",\n      \"action\": {\

[36;1m[1;3m[llm/end][0m [1m[1:chain:AgentExecutor > 3:tool:alphabet-earnings > 4:chain:RetrievalQA > 5:chain:StuffDocumentsChain > 6:chain:LLMChain > 7:llm:ChatOpenAI] [1.61s] Exiting LLM run with output:
[0m{
  "generations": [
    [
      {
        "text": "Alphabet's revenue for the quarter ended March 31, 2023, was $69,787 million.",
        "generation_info": null,
        "message": {
          "content": "Alphabet's revenue for the quarter ended March 31, 2023, was $69,787 million.",
          "additional_kwargs": {},
          "example": false
        }
      }
    ]
  ],
  "llm_output": {
    "token_usage": {
      "prompt_tokens": 2335,
      "completion_tokens": 23,
      "total_tokens": 2358
    },
    "model_name": "gpt-3.5-turbo-0613"
  },
  "run": null
}
[36;1m[1;3m[chain/end][0m [1m[1:chain:AgentExecutor > 3:tool:alphabet-earnings > 4:chain:RetrievalQA > 5:chain:StuffDocumentsChain > 6:chain:LLMChain] [1.61s] Exiting Chain run with output:
[0m{
  "text": "Alph

[36;1m[1;3m[llm/end][0m [1m[1:chain:AgentExecutor > 8:tool:tesla-earnings > 9:chain:RetrievalQA > 10:chain:StuffDocumentsChain > 11:chain:LLMChain > 12:llm:ChatOpenAI] [1.17s] Exiting LLM run with output:
[0m{
  "generations": [
    [
      {
        "text": "Tesla's revenue for Q1-2023 was $23.329 billion.",
        "generation_info": null,
        "message": {
          "content": "Tesla's revenue for Q1-2023 was $23.329 billion.",
          "additional_kwargs": {},
          "example": false
        }
      }
    ]
  ],
  "llm_output": {
    "token_usage": {
      "prompt_tokens": 2246,
      "completion_tokens": 16,
      "total_tokens": 2262
    },
    "model_name": "gpt-3.5-turbo-0613"
  },
  "run": null
}
[36;1m[1;3m[chain/end][0m [1m[1:chain:AgentExecutor > 8:tool:tesla-earnings > 9:chain:RetrievalQA > 10:chain:StuffDocumentsChain > 11:chain:LLMChain] [1.17s] Exiting Chain run with output:
[0m{
  "text": "Tesla's revenue for Q1-2023 was $23.329 billion."
}
[36;1m[1;

{'input': 'did alphabet or tesla have more revenue?',
 'output': 'Alphabet had a revenue of $69,787 million, while Tesla had a revenue of $23.329 billion. Therefore, Alphabet had more revenue than Tesla.'}