core: release 0.2.6 (#22868 )

core[patch]: Treat type as a special field when merging lists (#22750 )
Should we even log a warning? At least for Anthropic, it's expected to get e.g. `text_block` followed by `text_delta`. @ccurme @baskaryan @efriis
2026-02-16 01:59:52 +00:00 · 2024-06-13 22:22:34 +00:00 · 2024-06-13 15:08:24 -07:00 · 2024-06-13 15:02:48 -07:00 · 2024-06-13 20:29:34 +00:00 · 2024-06-13 16:13:15 -04:00
422 changed files with 23747 additions and 21174 deletions
--- a/.devcontainer/README.md
+++ b/.devcontainer/README.md
@@ -10,7 +10,7 @@ You can use the dev container configuration in this folder to build and run the
 You may use the button above, or follow these steps to open this repo in a Codespace:
 1. Click the **Code** drop-down menu at the top of https://github.com/langchain-ai/langchain.
 1. Click on the **Codespaces** tab.
-1. Click **Create codespace on master** .
+1. Click **Create codespace on master**.

 For more info, check out the [GitHub documentation](https://docs.github.com/en/free-pro-team@latest/github/developing-online-with-codespaces/creating-a-codespace#creating-a-codespace).
  
--- a/.github/workflows/_compile_integration_test.yml
+++ b/.github/workflows/_compile_integration_test.yml
@@ -24,6 +24,7 @@ jobs:
          - "3.9"
          - "3.10"
          - "3.11"
+          - "3.12"
    name: "poetry run pytest -m compile tests/integration_tests #${{ matrix.python-version }}"
    steps:
      - uses: actions/checkout@v4
--- a/.github/workflows/_dependencies.yml
+++ b/.github/workflows/_dependencies.yml
@@ -28,6 +28,7 @@ jobs:
          - "3.9"
          - "3.10"
          - "3.11"
+          - "3.12"
    name: dependency checks ${{ matrix.python-version }}
    steps:
      - uses: actions/checkout@v4
--- a/.github/workflows/_lint.yml
+++ b/.github/workflows/_lint.yml
@@ -34,7 +34,7 @@ jobs:
        # so linting on fewer versions makes CI faster.
        python-version:
          - "3.8"
-          - "3.11"
+          - "3.12"
    steps:
      - uses: actions/checkout@v4

--- a/.github/workflows/_test.yml
+++ b/.github/workflows/_test.yml
@@ -28,6 +28,7 @@ jobs:
          - "3.9"
          - "3.10"
          - "3.11"
+          - "3.12"
    name: "make test #${{ matrix.python-version }}"
    steps:
      - uses: actions/checkout@v4
--- a/.github/workflows/_test_doc_imports.yml
+++ b/.github/workflows/_test_doc_imports.yml
@@ -12,7 +12,7 @@ jobs:
    strategy:
      matrix:
        python-version:
-          - "3.11"
+          - "3.12"
    name: "check doc imports #${{ matrix.python-version }}"
    steps:
      - uses: actions/checkout@v4
--- a/.github/workflows/check-broken-links.yml
+++ b/.github/workflows/check-broken-links.yml
@@ -7,6 +7,7 @@ on:

 jobs:
  check-links:
+    if: github.repository_owner == 'langchain-ai'
    runs-on: ubuntu-latest
    steps:
      - uses: actions/checkout@v4
--- a/.github/workflows/check_diffs.yml
+++ b/.github/workflows/check_diffs.yml
@@ -104,6 +104,7 @@ jobs:
          - "3.9"
          - "3.10"
          - "3.11"
+          - "3.12"
    runs-on: ubuntu-latest
    defaults:
      run:
@@ -123,7 +124,9 @@ jobs:
        shell: bash
        run: |
          echo "Running extended tests, installing dependencies with poetry..."
-          poetry install -E extended_testing --with test
+          poetry install --with test
+          poetry run pip install uv
+          poetry run uv pip install -r extended_testing_deps.txt

      - name: Run extended tests
        run: make extended_tests
--- a/.github/workflows/check_new_docs.yml
+++ b/.github/workflows/check_new_docs.yml
@@ -0,0 +1,31 @@
+---
+name: Integration docs lint
+
+on:
+  push:
+    branches: [master]
+  pull_request:
+
+# If another push to the same PR or branch happens while this workflow is still running,
+# cancel the earlier run in favor of the next run.
+#
+# There's no point in testing an outdated version of the code. GitHub only allows
+# a limited number of job runners to be active at the same time, so it's better to cancel
+# pointless jobs early so that more useful jobs can run sooner.
+concurrency:
+  group: ${{ github.workflow }}-${{ github.ref }}
+  cancel-in-progress: true
+
+jobs:
+  build:
+    runs-on: ubuntu-latest
+    steps:
+      - uses: actions/checkout@v4
+      - uses: actions/setup-python@v5
+        with:
+          python-version: '3.10'
+      - id: files
+        uses: Ana06/get-changed-files@v2.2.0
+      - name: Check new docs
+        run: |
+          python docs/scripts/check_templates.py ${{ steps.files.outputs.added }}
--- a/.github/workflows/scheduled_test.yml
+++ b/.github/workflows/scheduled_test.yml
@@ -10,6 +10,7 @@ env:

 jobs:
  build:
+    if: github.repository_owner == 'langchain-ai'
    name: Python ${{ matrix.python-version }} - ${{ matrix.working-directory }}
    runs-on: ubuntu-latest
    strategy:
--- a/.gitignore
+++ b/.gitignore
@@ -133,6 +133,7 @@ env.bak/

 # mypy
 .mypy_cache/
+.mypy_cache_test/
 .dmypy.json
 dmypy.json

--- a/README.md
+++ b/README.md
@@ -106,7 +106,7 @@ Retrieval Augmented Generation involves [loading data](https://python.langchain.

 **🤖 Agents**

-Agents allow an LLM autonomy over how a task is accomplished. Agents make decisions about which Actions to take, then take that Action, observe the result, and repeat until the task is complete done. LangChain provides a [standard interface for agents](https://python.langchain.com/v0.2/docs/concepts/#agents) along with the [LangGraph](https://github.com/langchain-ai/langgraph) extension for building custom agents.
+Agents allow an LLM autonomy over how a task is accomplished. Agents make decisions about which Actions to take, then take that Action, observe the result, and repeat until the task is complete. LangChain provides a [standard interface for agents](https://python.langchain.com/v0.2/docs/concepts/#agents) along with the [LangGraph](https://github.com/langchain-ai/langgraph) extension for building custom agents.

 ## 📖 Documentation

--- a/cookbook/autogpt/marathon_times.ipynb
+++ b/cookbook/autogpt/marathon_times.ipynb
@@ -46,7 +46,7 @@
    "from langchain_experimental.autonomous_agents import AutoGPT\n",
    "from langchain_openai import ChatOpenAI\n",
    "\n",
-    "# Needed synce jupyter runs an async eventloop\n",
+    "# Needed since jupyter runs an async eventloop\n",
    "nest_asyncio.apply()"
   ]
  },
--- a/cookbook/azure_container_apps_dynamic_sessions_data_analyst.ipynb
+++ b/cookbook/azure_container_apps_dynamic_sessions_data_analyst.ipynb
@@ -273,7 +273,7 @@
   "source": [
    "# Tool schema for querying SQL db\n",
    "class create_df_from_sql(BaseModel):\n",
-    "    \"\"\"Execute a PostgreSQL SELECT statement and use the results to create a DataFrame with the given colum names.\"\"\"\n",
+    "    \"\"\"Execute a PostgreSQL SELECT statement and use the results to create a DataFrame with the given column names.\"\"\"\n",
    "\n",
    "    select_query: str = Field(..., description=\"A PostgreSQL SELECT statement.\")\n",
    "    # We're going to convert the results to a Pandas DataFrame that we pass\n",
--- a/cookbook/nomic_multimodal_rag.ipynb
+++ b/cookbook/nomic_multimodal_rag.ipynb
@@ -0,0 +1,497 @@
+{
+ "cells": [
+  {
+   "attachments": {},
+   "cell_type": "markdown",
+   "id": "9fc3897d-176f-4729-8fd1-cfb4add53abd",
+   "metadata": {},
+   "source": [
+    "## Nomic multi-modal RAG\n",
+    "\n",
+    "Many documents contain a mixture of content types, including text and images. \n",
+    "\n",
+    "Yet, information captured in images is lost in most RAG applications.\n",
+    "\n",
+    "With the emergence of multimodal LLMs, like [GPT-4V](https://openai.com/research/gpt-4v-system-card), it is worth considering how to utilize images in RAG:\n",
+    "\n",
+    "In this demo we\n",
+    "\n",
+    "* Use multimodal embeddings from Nomic Embed [Vision](https://huggingface.co/nomic-ai/nomic-embed-vision-v1.5) and [Text](https://huggingface.co/nomic-ai/nomic-embed-text-v1.5) to embed images and text\n",
+    "* Retrieve both using similarity search\n",
+    "* Pass raw images and text chunks to a multimodal LLM for answer synthesis \n",
+    "\n",
+    "## Signup\n",
+    "\n",
+    "Get your API token, then run:\n",
+    "```\n",
+    "! nomic login\n",
+    "```\n",
+    "\n",
+    "Then run with your generated API token \n",
+    "```\n",
+    "! nomic login < token > \n",
+    "```\n",
+    "\n",
+    "## Packages\n",
+    "\n",
+    "For `unstructured`, you will also need `poppler` ([installation instructions](https://pdf2image.readthedocs.io/en/latest/installation.html)) and `tesseract` ([installation instructions](https://tesseract-ocr.github.io/tessdoc/Installation.html)) in your system."
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "54926b9b-75c2-4cd4-8f14-b3882a0d370b",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "! nomic login token"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "febbc459-ebba-4c1a-a52b-fed7731593f8",
+   "metadata": {
+    "scrolled": true
+   },
+   "outputs": [],
+   "source": [
+    "! pip install -U langchain-nomic langchain_community tiktoken langchain-openai chromadb langchain # (newest versions required for multi-modal)"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "acbdc603-39e2-4a5f-836c-2bbaecd46b0b",
+   "metadata": {
+    "scrolled": true
+   },
+   "outputs": [],
+   "source": [
+    "# lock to 0.10.19 due to a persistent bug in more recent versions\n",
+    "! pip install \"unstructured[all-docs]==0.10.19\" pillow pydantic lxml pillow matplotlib tiktoken"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "1e94b3fb-8e3e-4736-be0a-ad881626c7bd",
+   "metadata": {},
+   "source": [
+    "## Data Loading\n",
+    "\n",
+    "### Partition PDF text and images\n",
+    "  \n",
+    "Let's look at an example pdfs containing interesting images.\n",
+    "\n",
+    "1/ Art from the J Paul Getty museum:\n",
+    "\n",
+    " * Here is a [zip file](https://drive.google.com/file/d/18kRKbq2dqAhhJ3DfZRnYcTBEUfYxe1YR/view?usp=sharing) with the PDF and the already extracted images. \n",
+    "* https://www.getty.edu/publications/resources/virtuallibrary/0892360224.pdf\n",
+    "\n",
+    "2/ Famous photographs from library of congress:\n",
+    "\n",
+    "* https://www.loc.gov/lcm/pdf/LCM_2020_1112.pdf\n",
+    "* We'll use this as an example below\n",
+    "\n",
+    "We can use `partition_pdf` below from [Unstructured](https://unstructured-io.github.io/unstructured/introduction.html#key-concepts) to extract text and images.\n",
+    "\n",
+    "To supply this to extract the images:\n",
+    "```\n",
+    "extract_images_in_pdf=True\n",
+    "```\n",
+    "\n",
+    "\n",
+    "\n",
+    "If using this zip file, then you can simply process the text only with:\n",
+    "```\n",
+    "extract_images_in_pdf=False\n",
+    "```"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "9646b524-71a7-4b2a-bdc8-0b81f77e968f",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "# Folder with pdf and extracted images\n",
+    "from pathlib import Path\n",
+    "\n",
+    "# replace with actual path to images\n",
+    "path = Path(\"../art\")"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "77f096ab-a933-41d0-8f4e-1efc83998fc3",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "path.resolve()"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "bc4839c0-8773-4a07-ba59-5364501269b2",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "# Extract images, tables, and chunk text\n",
+    "from unstructured.partition.pdf import partition_pdf\n",
+    "\n",
+    "raw_pdf_elements = partition_pdf(\n",
+    "    filename=str(path.resolve()) + \"/getty.pdf\",\n",
+    "    extract_images_in_pdf=False,\n",
+    "    infer_table_structure=True,\n",
+    "    chunking_strategy=\"by_title\",\n",
+    "    max_characters=4000,\n",
+    "    new_after_n_chars=3800,\n",
+    "    combine_text_under_n_chars=2000,\n",
+    "    image_output_dir_path=path,\n",
+    ")"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "969545ad",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "# Categorize text elements by type\n",
+    "tables = []\n",
+    "texts = []\n",
+    "for element in raw_pdf_elements:\n",
+    "    if \"unstructured.documents.elements.Table\" in str(type(element)):\n",
+    "        tables.append(str(element))\n",
+    "    elif \"unstructured.documents.elements.CompositeElement\" in str(type(element)):\n",
+    "        texts.append(str(element))"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "5d8e6349-1547-4cbf-9c6f-491d8610ec10",
+   "metadata": {},
+   "source": [
+    "## Multi-modal embeddings with our document\n",
+    "\n",
+    "We will use [nomic-embed-vision-v1.5](https://huggingface.co/nomic-ai/nomic-embed-vision-v1.5) embeddings. This model is aligned \n",
+    "to [nomic-embed-text-v1.5](https://huggingface.co/nomic-ai/nomic-embed-text-v1.5) allowing for multimodal semantic search and Multimodal RAG!"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "4bc15842-cb95-4f84-9eb5-656b0282a800",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "import os\n",
+    "import uuid\n",
+    "\n",
+    "import chromadb\n",
+    "import numpy as np\n",
+    "from langchain_community.vectorstores import Chroma\n",
+    "from langchain_nomic import NomicEmbeddings\n",
+    "from PIL import Image as _PILImage\n",
+    "\n",
+    "# Create chroma\n",
+    "text_vectorstore = Chroma(\n",
+    "    collection_name=\"mm_rag_clip_photos_text\",\n",
+    "    embedding_function=NomicEmbeddings(\n",
+    "        vision_model=\"nomic-embed-vision-v1.5\", model=\"nomic-embed-text-v1.5\"\n",
+    "    ),\n",
+    ")\n",
+    "image_vectorstore = Chroma(\n",
+    "    collection_name=\"mm_rag_clip_photos_image\",\n",
+    "    embedding_function=NomicEmbeddings(\n",
+    "        vision_model=\"nomic-embed-vision-v1.5\", model=\"nomic-embed-text-v1.5\"\n",
+    "    ),\n",
+    ")\n",
+    "\n",
+    "# Get image URIs with .jpg extension only\n",
+    "image_uris = sorted(\n",
+    "    [\n",
+    "        os.path.join(path, image_name)\n",
+    "        for image_name in os.listdir(path)\n",
+    "        if image_name.endswith(\".jpg\")\n",
+    "    ]\n",
+    ")\n",
+    "\n",
+    "# Add images\n",
+    "image_vectorstore.add_images(uris=image_uris)\n",
+    "\n",
+    "# Add documents\n",
+    "text_vectorstore.add_texts(texts=texts)\n",
+    "\n",
+    "# Make retriever\n",
+    "image_retriever = image_vectorstore.as_retriever()\n",
+    "text_retriever = text_vectorstore.as_retriever()"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "02a186d0-27e0-4820-8092-63b5349dd25d",
+   "metadata": {},
+   "source": [
+    "## RAG\n",
+    "\n",
+    "`vectorstore.add_images` will store / retrieve images as base64 encoded strings.\n",
+    "\n",
+    "These can be passed to [GPT-4V](https://platform.openai.com/docs/guides/vision)."
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "344f56a8-0dc3-433e-851c-3f7600c7a72b",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "import base64\n",
+    "import io\n",
+    "from io import BytesIO\n",
+    "\n",
+    "import numpy as np\n",
+    "from PIL import Image\n",
+    "\n",
+    "\n",
+    "def resize_base64_image(base64_string, size=(128, 128)):\n",
+    "    \"\"\"\n",
+    "    Resize an image encoded as a Base64 string.\n",
+    "\n",
+    "    Args:\n",
+    "    base64_string (str): Base64 string of the original image.\n",
+    "    size (tuple): Desired size of the image as (width, height).\n",
+    "\n",
+    "    Returns:\n",
+    "    str: Base64 string of the resized image.\n",
+    "    \"\"\"\n",
+    "    # Decode the Base64 string\n",
+    "    img_data = base64.b64decode(base64_string)\n",
+    "    img = Image.open(io.BytesIO(img_data))\n",
+    "\n",
+    "    # Resize the image\n",
+    "    resized_img = img.resize(size, Image.LANCZOS)\n",
+    "\n",
+    "    # Save the resized image to a bytes buffer\n",
+    "    buffered = io.BytesIO()\n",
+    "    resized_img.save(buffered, format=img.format)\n",
+    "\n",
+    "    # Encode the resized image to Base64\n",
+    "    return base64.b64encode(buffered.getvalue()).decode(\"utf-8\")\n",
+    "\n",
+    "\n",
+    "def is_base64(s):\n",
+    "    \"\"\"Check if a string is Base64 encoded\"\"\"\n",
+    "    try:\n",
+    "        return base64.b64encode(base64.b64decode(s)) == s.encode()\n",
+    "    except Exception:\n",
+    "        return False\n",
+    "\n",
+    "\n",
+    "def split_image_text_types(docs):\n",
+    "    \"\"\"Split numpy array images and texts\"\"\"\n",
+    "    images = []\n",
+    "    text = []\n",
+    "    for doc in docs:\n",
+    "        doc = doc.page_content  # Extract Document contents\n",
+    "        if is_base64(doc):\n",
+    "            # Resize image to avoid OAI server error\n",
+    "            images.append(\n",
+    "                resize_base64_image(doc, size=(250, 250))\n",
+    "            )  # base64 encoded str\n",
+    "        else:\n",
+    "            text.append(doc)\n",
+    "    return {\"images\": images, \"texts\": text}"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "23a2c1d8-fea6-4152-b184-3172dd46c735",
+   "metadata": {},
+   "source": [
+    "Currently, we format the inputs using a `RunnableLambda` while we add image support to `ChatPromptTemplates`.\n",
+    "\n",
+    "Our runnable follows the classic RAG flow - \n",
+    "\n",
+    "* We first compute the context (both \"texts\" and \"images\" in this case) and the question (just a RunnablePassthrough here) \n",
+    "* Then we pass this into our prompt template, which is a custom function that formats the message for the gpt-4-vision-preview model. \n",
+    "* And finally we parse the output as a string."
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "5d8919dc-c238-4746-86ba-45d940a7d260",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "import os\n",
+    "\n",
+    "os.environ[\"OPENAI_API_KEY\"] = \"\""
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "4c93fab3-74c4-4f1d-958a-0bc4cdd0797e",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "from operator import itemgetter\n",
+    "\n",
+    "from langchain_core.messages import HumanMessage, SystemMessage\n",
+    "from langchain_core.output_parsers import StrOutputParser\n",
+    "from langchain_core.runnables import RunnableLambda, RunnablePassthrough\n",
+    "from langchain_openai import ChatOpenAI\n",
+    "\n",
+    "\n",
+    "def prompt_func(data_dict):\n",
+    "    # Joining the context texts into a single string\n",
+    "    formatted_texts = \"\\n\".join(data_dict[\"text_context\"][\"texts\"])\n",
+    "    messages = []\n",
+    "\n",
+    "    # Adding image(s) to the messages if present\n",
+    "    if data_dict[\"image_context\"][\"images\"]:\n",
+    "        image_message = {\n",
+    "            \"type\": \"image_url\",\n",
+    "            \"image_url\": {\n",
+    "                \"url\": f\"data:image/jpeg;base64,{data_dict['image_context']['images'][0]}\"\n",
+    "            },\n",
+    "        }\n",
+    "        messages.append(image_message)\n",
+    "\n",
+    "    # Adding the text message for analysis\n",
+    "    text_message = {\n",
+    "        \"type\": \"text\",\n",
+    "        \"text\": (\n",
+    "            \"As an expert art critic and historian, your task is to analyze and interpret images, \"\n",
+    "            \"considering their historical and cultural significance. Alongside the images, you will be \"\n",
+    "            \"provided with related text to offer context. Both will be retrieved from a vectorstore based \"\n",
+    "            \"on user-input keywords. Please use your extensive knowledge and analytical skills to provide a \"\n",
+    "            \"comprehensive summary that includes:\\n\"\n",
+    "            \"- A detailed description of the visual elements in the image.\\n\"\n",
+    "            \"- The historical and cultural context of the image.\\n\"\n",
+    "            \"- An interpretation of the image's symbolism and meaning.\\n\"\n",
+    "            \"- Connections between the image and the related text.\\n\\n\"\n",
+    "            f\"User-provided keywords: {data_dict['question']}\\n\\n\"\n",
+    "            \"Text and / or tables:\\n\"\n",
+    "            f\"{formatted_texts}\"\n",
+    "        ),\n",
+    "    }\n",
+    "    messages.append(text_message)\n",
+    "\n",
+    "    return [HumanMessage(content=messages)]\n",
+    "\n",
+    "\n",
+    "model = ChatOpenAI(temperature=0, model=\"gpt-4-vision-preview\", max_tokens=1024)\n",
+    "\n",
+    "# RAG pipeline\n",
+    "chain = (\n",
+    "    {\n",
+    "        \"text_context\": text_retriever | RunnableLambda(split_image_text_types),\n",
+    "        \"image_context\": image_retriever | RunnableLambda(split_image_text_types),\n",
+    "        \"question\": RunnablePassthrough(),\n",
+    "    }\n",
+    "    | RunnableLambda(prompt_func)\n",
+    "    | model\n",
+    "    | StrOutputParser()\n",
+    ")"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "1566096d-97c2-4ddc-ba4a-6ef88c525e4e",
+   "metadata": {},
+   "source": [
+    "## Test retrieval and run RAG"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "90121e56-674b-473b-871d-6e4753fd0c45",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "from IPython.display import HTML, display\n",
+    "\n",
+    "\n",
+    "def plt_img_base64(img_base64):\n",
+    "    # Create an HTML img tag with the base64 string as the source\n",
+    "    image_html = f'<img src=\"data:image/jpeg;base64,{img_base64}\" />'\n",
+    "\n",
+    "    # Display the image by rendering the HTML\n",
+    "    display(HTML(image_html))\n",
+    "\n",
+    "\n",
+    "docs = text_retriever.invoke(\"Women with children\", k=5)\n",
+    "for doc in docs:\n",
+    "    if is_base64(doc.page_content):\n",
+    "        plt_img_base64(doc.page_content)\n",
+    "    else:\n",
+    "        print(doc.page_content)"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "44eaa532-f035-4c04-b578-02339d42554c",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "docs = image_retriever.invoke(\"Women with children\", k=5)\n",
+    "for doc in docs:\n",
+    "    if is_base64(doc.page_content):\n",
+    "        plt_img_base64(doc.page_content)\n",
+    "    else:\n",
+    "        print(doc.page_content)"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "69fb15fd-76fc-49b4-806d-c4db2990027d",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "chain.invoke(\"Women with children\")"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "227f08b8-e732-4089-b65c-6eb6f9e48f15",
+   "metadata": {},
+   "source": [
+    "We can see the images retrieved in the LangSmith trace:\n",
+    "\n",
+    "LangSmith [trace](https://smith.langchain.com/public/69c558a5-49dc-4c60-a49b-3adbb70f74c5/r/e872c2c8-528c-468f-aefd-8b5cd730a673)."
+   ]
+  }
+ ],
+ "metadata": {
+  "kernelspec": {
+   "display_name": "Python 3 (ipykernel)",
+   "language": "python",
+   "name": "python3"
+  },
+  "language_info": {
+   "codemirror_mode": {
+    "name": "ipython",
+    "version": 3
+   },
+   "file_extension": ".py",
+   "mimetype": "text/x-python",
+   "name": "python",
+   "nbconvert_exporter": "python",
+   "pygments_lexer": "ipython3",
+   "version": "3.11.9"
+  }
+ },
+ "nbformat": 4,
+ "nbformat_minor": 5
+}
--- a/cookbook/oracleai_demo.ipynb
+++ b/cookbook/oracleai_demo.ipynb
@@ -86,8 +86,7 @@
    "\n",
    "import oracledb\n",
    "\n",
-    "# please update with your username, password, hostname and service_name\n",
-    "# please make sure this user has sufficient privileges to perform all below\n",
+    "# Update with your username, password, hostname, and service_name\n",
    "username = \"\"\n",
    "password = \"\"\n",
    "dsn = \"\"\n",
@@ -97,40 +96,45 @@
    "    print(\"Connection successful!\")\n",
    "\n",
    "    cursor = conn.cursor()\n",
-    "    cursor.execute(\n",
-    "        \"\"\"\n",
-    "    begin\n",
-    "        -- drop user\n",
-    "        begin\n",
-    "            execute immediate 'drop user testuser cascade';\n",
-    "        exception\n",
-    "            when others then\n",
-    "                dbms_output.put_line('Error setting up user.');\n",
-    "        end;\n",
-    "        execute immediate 'create user testuser identified by testuser';\n",
-    "        execute immediate 'grant connect, unlimited tablespace, create credential, create procedure, create any index to testuser';\n",
-    "        execute immediate 'create or replace directory DEMO_PY_DIR as ''/scratch/hroy/view_storage/hroy_devstorage/demo/orachain''';\n",
-    "        execute immediate 'grant read, write on directory DEMO_PY_DIR to public';\n",
-    "        execute immediate 'grant create mining model to testuser';\n",
-    "\n",
-    "        -- network access\n",
-    "        begin\n",
-    "            DBMS_NETWORK_ACL_ADMIN.APPEND_HOST_ACE(\n",
-    "                host => '*',\n",
-    "                ace => xs$ace_type(privilege_list => xs$name_list('connect'),\n",
-    "                                principal_name => 'testuser',\n",
-    "                                principal_type => xs_acl.ptype_db));\n",
-    "        end;\n",
-    "    end;\n",
-    "    \"\"\"\n",
-    "    )\n",
-    "    print(\"User setup done!\")\n",
-    "    cursor.close()\n",
+    "    try:\n",
+    "        cursor.execute(\n",
+    "            \"\"\"\n",
+    "            begin\n",
+    "                -- Drop user\n",
+    "                begin\n",
+    "                    execute immediate 'drop user testuser cascade';\n",
+    "                exception\n",
+    "                    when others then\n",
+    "                        dbms_output.put_line('Error dropping user: ' || SQLERRM);\n",
+    "                end;\n",
+    "                \n",
+    "                -- Create user and grant privileges\n",
+    "                execute immediate 'create user testuser identified by testuser';\n",
+    "                execute immediate 'grant connect, unlimited tablespace, create credential, create procedure, create any index to testuser';\n",
+    "                execute immediate 'create or replace directory DEMO_PY_DIR as ''/scratch/hroy/view_storage/hroy_devstorage/demo/orachain''';\n",
+    "                execute immediate 'grant read, write on directory DEMO_PY_DIR to public';\n",
+    "                execute immediate 'grant create mining model to testuser';\n",
+    "                \n",
+    "                -- Network access\n",
+    "                begin\n",
+    "                    DBMS_NETWORK_ACL_ADMIN.APPEND_HOST_ACE(\n",
+    "                        host => '*',\n",
+    "                        ace => xs$ace_type(privilege_list => xs$name_list('connect'),\n",
+    "                                           principal_name => 'testuser',\n",
+    "                                           principal_type => xs_acl.ptype_db)\n",
+    "                    );\n",
+    "                end;\n",
+    "            end;\n",
+    "            \"\"\"\n",
+    "        )\n",
+    "        print(\"User setup done!\")\n",
+    "    except Exception as e:\n",
+    "        print(f\"User setup failed with error: {e}\")\n",
+    "    finally:\n",
+    "        cursor.close()\n",
    "    conn.close()\n",
    "except Exception as e:\n",
-    "    print(\"User setup failed!\")\n",
-    "    cursor.close()\n",
-    "    conn.close()\n",
+    "    print(f\"Connection failed with error: {e}\")\n",
    "    sys.exit(1)"
   ]
  },
--- a/docs/api_reference/create_api_rst.py
+++ b/docs/api_reference/create_api_rst.py
@@ -128,11 +128,11 @@ def _load_package_modules(
    of the modules/packages are part of the package vs. 3rd party or built-in.

    Parameters:
-        package_directory: Path to the package directory.
-        submodule: Optional name of submodule to load.
+        package_directory (Union[str, Path]): Path to the package directory.
+        submodule (Optional[str]): Optional name of submodule to load.

    Returns:
-        list: A list of loaded module objects.
+        Dict[str, ModuleMembers]: A dictionary where keys are module names and values are ModuleMembers objects.
    """
    package_path = (
        Path(package_directory)
--- a/docs/api_reference/guide_imports.json
+++ b/docs/api_reference/guide_imports.json
--- a/docs/docs/additional_resources/arxiv_references.mdx
+++ b/docs/docs/additional_resources/arxiv_references.mdx
@@ -4,6 +4,9 @@ LangChain implements the latest research in the field of Natural Language Proces
 This page contains `arXiv` papers referenced in the LangChain Documentation, API Reference,
 Templates, and Cookbooks.

+From the opposite direction, scientists use LangChain in research and reference LangChain in the research papers. 
+Here you find [such papers](https://arxiv.org/search/?query=langchain&searchtype=all&source=header).
+
 ## Summary

 | arXiv id / Title | Authors | Published date 🔻 | LangChain Documentation|
@@ -21,21 +24,21 @@ This page contains `arXiv` papers referenced in the LangChain Documentation, API
 | `2305.08291v1` [Large Language Model Guided Tree-of-Thought](http://arxiv.org/abs/2305.08291v1) | Jieyi Long | 2023-05-15 | `API:` [langchain_experimental.tot](https://api.python.langchain.com/en/latest/experimental_api_reference.html#module-langchain_experimental.tot), `Cookbook:` [tree_of_thought](https://github.com/langchain-ai/langchain/blob/master/cookbook/tree_of_thought.ipynb)
 | `2305.04091v3` [Plan-and-Solve Prompting: Improving Zero-Shot Chain-of-Thought Reasoning by Large Language Models](http://arxiv.org/abs/2305.04091v3) | Lei Wang, Wanyu Xu, Yihuai Lan,  et al. | 2023-05-06 | `Cookbook:` [plan_and_execute_agent](https://github.com/langchain-ai/langchain/blob/master/cookbook/plan_and_execute_agent.ipynb)
 | `2304.08485v2` [Visual Instruction Tuning](http://arxiv.org/abs/2304.08485v2) | Haotian Liu, Chunyuan Li, Qingyang Wu,  et al. | 2023-04-17 | `Cookbook:` [Semi_structured_and_multi_modal_RAG](https://github.com/langchain-ai/langchain/blob/master/cookbook/Semi_structured_and_multi_modal_RAG.ipynb), [Semi_structured_multi_modal_RAG_LLaMA2](https://github.com/langchain-ai/langchain/blob/master/cookbook/Semi_structured_multi_modal_RAG_LLaMA2.ipynb)
-| `2304.03442v2` [Generative Agents: Interactive Simulacra of Human Behavior](http://arxiv.org/abs/2304.03442v2) | Joon Sung Park, Joseph C. O'Brien, Carrie J. Cai,  et al. | 2023-04-07 | `Cookbook:` [multiagent_bidding](https://github.com/langchain-ai/langchain/blob/master/cookbook/multiagent_bidding.ipynb), [generative_agents_interactive_simulacra_of_human_behavior](https://github.com/langchain-ai/langchain/blob/master/cookbook/generative_agents_interactive_simulacra_of_human_behavior.ipynb)
+| `2304.03442v2` [Generative Agents: Interactive Simulacra of Human Behavior](http://arxiv.org/abs/2304.03442v2) | Joon Sung Park, Joseph C. O'Brien, Carrie J. Cai,  et al. | 2023-04-07 | `Cookbook:` [generative_agents_interactive_simulacra_of_human_behavior](https://github.com/langchain-ai/langchain/blob/master/cookbook/generative_agents_interactive_simulacra_of_human_behavior.ipynb), [multiagent_bidding](https://github.com/langchain-ai/langchain/blob/master/cookbook/multiagent_bidding.ipynb)
 | `2303.17760v2` [CAMEL: Communicative Agents for "Mind" Exploration of Large Language Model Society](http://arxiv.org/abs/2303.17760v2) | Guohao Li, Hasan Abed Al Kader Hammoud, Hani Itani,  et al. | 2023-03-31 | `Cookbook:` [camel_role_playing](https://github.com/langchain-ai/langchain/blob/master/cookbook/camel_role_playing.ipynb)
 | `2303.17580v4` [HuggingGPT: Solving AI Tasks with ChatGPT and its Friends in Hugging Face](http://arxiv.org/abs/2303.17580v4) | Yongliang Shen, Kaitao Song, Xu Tan,  et al. | 2023-03-30 | `API:` [langchain_experimental.autonomous_agents](https://api.python.langchain.com/en/latest/experimental_api_reference.html#module-langchain_experimental.autonomous_agents), `Cookbook:` [hugginggpt](https://github.com/langchain-ai/langchain/blob/master/cookbook/hugginggpt.ipynb)
 | `2303.08774v6` [GPT-4 Technical Report](http://arxiv.org/abs/2303.08774v6) | OpenAI, Josh Achiam, Steven Adler,  et al. | 2023-03-15 | `Docs:` [docs/integrations/vectorstores/mongodb_atlas](https://python.langchain.com/docs/integrations/vectorstores/mongodb_atlas)
-| `2301.10226v4` [A Watermark for Large Language Models](http://arxiv.org/abs/2301.10226v4) | John Kirchenbauer, Jonas Geiping, Yuxin Wen,  et al. | 2023-01-24 | `API:` [langchain_community.llms...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_community.llms...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_huggingface.llms...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_community.llms...OCIModelDeploymentTGI](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.oci_data_science_model_deployment_endpoint.OCIModelDeploymentTGI.html#langchain_community.llms.oci_data_science_model_deployment_endpoint.OCIModelDeploymentTGI)
-| `2212.10496v1` [Precise Zero-Shot Dense Retrieval without Relevance Labels](http://arxiv.org/abs/2212.10496v1) | Luyu Gao, Xueguang Ma, Jimmy Lin,  et al. | 2022-12-20 | `API:` [langchain.chains...HypotheticalDocumentEmbedder](https://api.python.langchain.com/en/latest/chains/langchain.chains.hyde.base.HypotheticalDocumentEmbedder.html#langchain.chains.hyde.base.HypotheticalDocumentEmbedder), `Template:` [hyde](https://python.langchain.com/docs/templates/hyde), `Cookbook:` [hypothetical_document_embeddings](https://github.com/langchain-ai/langchain/blob/master/cookbook/hypothetical_document_embeddings.ipynb)
+| `2301.10226v4` [A Watermark for Large Language Models](http://arxiv.org/abs/2301.10226v4) | John Kirchenbauer, Jonas Geiping, Yuxin Wen,  et al. | 2023-01-24 | `API:` [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_community...OCIModelDeploymentTGI](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.oci_data_science_model_deployment_endpoint.OCIModelDeploymentTGI.html#langchain_community.llms.oci_data_science_model_deployment_endpoint.OCIModelDeploymentTGI), [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint)
+| `2212.10496v1` [Precise Zero-Shot Dense Retrieval without Relevance Labels](http://arxiv.org/abs/2212.10496v1) | Luyu Gao, Xueguang Ma, Jimmy Lin,  et al. | 2022-12-20 | `API:` [langchain...HypotheticalDocumentEmbedder](https://api.python.langchain.com/en/latest/chains/langchain.chains.hyde.base.HypotheticalDocumentEmbedder.html#langchain.chains.hyde.base.HypotheticalDocumentEmbedder), `Template:` [hyde](https://python.langchain.com/docs/templates/hyde), `Cookbook:` [hypothetical_document_embeddings](https://github.com/langchain-ai/langchain/blob/master/cookbook/hypothetical_document_embeddings.ipynb)
 | `2212.07425v3` [Robust and Explainable Identification of Logical Fallacies in Natural Language Arguments](http://arxiv.org/abs/2212.07425v3) | Zhivar Sourati, Vishnu Priya Prasanna Venkatesh, Darshan Deshpande,  et al. | 2022-12-12 | `API:` [langchain_experimental.fallacy_removal](https://api.python.langchain.com/en/latest/experimental_api_reference.html#module-langchain_experimental.fallacy_removal)
-| `2211.13892v2` [Complementary Explanations for Effective In-Context Learning](http://arxiv.org/abs/2211.13892v2) | Xi Ye, Srinivasan Iyer, Asli Celikyilmaz,  et al. | 2022-11-25 | `API:` [langchain_core.example_selectors...MaxMarginalRelevanceExampleSelector](https://api.python.langchain.com/en/latest/example_selectors/langchain_core.example_selectors.semantic_similarity.MaxMarginalRelevanceExampleSelector.html#langchain_core.example_selectors.semantic_similarity.MaxMarginalRelevanceExampleSelector)
-| `2211.10435v2` [PAL: Program-aided Language Models](http://arxiv.org/abs/2211.10435v2) | Luyu Gao, Aman Madaan, Shuyan Zhou,  et al. | 2022-11-18 | `API:` [langchain_experimental.pal_chain](https://api.python.langchain.com/en/latest/experimental_api_reference.html#module-langchain_experimental.pal_chain), [langchain_experimental.pal_chain...PALChain](https://api.python.langchain.com/en/latest/pal_chain/langchain_experimental.pal_chain.base.PALChain.html#langchain_experimental.pal_chain.base.PALChain), `Cookbook:` [program_aided_language_model](https://github.com/langchain-ai/langchain/blob/master/cookbook/program_aided_language_model.ipynb)
+| `2211.13892v2` [Complementary Explanations for Effective In-Context Learning](http://arxiv.org/abs/2211.13892v2) | Xi Ye, Srinivasan Iyer, Asli Celikyilmaz,  et al. | 2022-11-25 | `API:` [langchain_core...MaxMarginalRelevanceExampleSelector](https://api.python.langchain.com/en/latest/example_selectors/langchain_core.example_selectors.semantic_similarity.MaxMarginalRelevanceExampleSelector.html#langchain_core.example_selectors.semantic_similarity.MaxMarginalRelevanceExampleSelector)
+| `2211.10435v2` [PAL: Program-aided Language Models](http://arxiv.org/abs/2211.10435v2) | Luyu Gao, Aman Madaan, Shuyan Zhou,  et al. | 2022-11-18 | `API:` [langchain_experimental...PALChain](https://api.python.langchain.com/en/latest/pal_chain/langchain_experimental.pal_chain.base.PALChain.html#langchain_experimental.pal_chain.base.PALChain), [langchain_experimental.pal_chain](https://api.python.langchain.com/en/latest/experimental_api_reference.html#module-langchain_experimental.pal_chain), `Cookbook:` [program_aided_language_model](https://github.com/langchain-ai/langchain/blob/master/cookbook/program_aided_language_model.ipynb)
 | `2209.10785v2` [Deep Lake: a Lakehouse for Deep Learning](http://arxiv.org/abs/2209.10785v2) | Sasun Hambardzumyan, Abhinav Tuli, Levon Ghukasyan,  et al. | 2022-09-22 | `Docs:` [docs/integrations/providers/activeloop_deeplake](https://python.langchain.com/docs/integrations/providers/activeloop_deeplake)
-| `2205.12654v1` [Bitext Mining Using Distilled Sentence Representations for Low-Resource Languages](http://arxiv.org/abs/2205.12654v1) | Kevin Heffernan, Onur Çelebi, Holger Schwenk | 2022-05-25 | `API:` [langchain_community.embeddings...LaserEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.laser.LaserEmbeddings.html#langchain_community.embeddings.laser.LaserEmbeddings)
-| `2204.00498v1` [Evaluating the Text-to-SQL Capabilities of Large Language Models](http://arxiv.org/abs/2204.00498v1) | Nitarshan Rajkumar, Raymond Li, Dzmitry Bahdanau | 2022-03-15 | `API:` [langchain_community.utilities...SparkSQL](https://api.python.langchain.com/en/latest/utilities/langchain_community.utilities.spark_sql.SparkSQL.html#langchain_community.utilities.spark_sql.SparkSQL), [langchain_community.utilities...SQLDatabase](https://api.python.langchain.com/en/latest/utilities/langchain_community.utilities.sql_database.SQLDatabase.html#langchain_community.utilities.sql_database.SQLDatabase)
-| `2202.00666v5` [Locally Typical Sampling](http://arxiv.org/abs/2202.00666v5) | Clara Meister, Tiago Pimentel, Gian Wiher,  et al. | 2022-02-01 | `API:` [langchain_community.llms...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_community.llms...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_huggingface.llms...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint)
+| `2205.12654v1` [Bitext Mining Using Distilled Sentence Representations for Low-Resource Languages](http://arxiv.org/abs/2205.12654v1) | Kevin Heffernan, Onur Çelebi, Holger Schwenk | 2022-05-25 | `API:` [langchain_community...LaserEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.laser.LaserEmbeddings.html#langchain_community.embeddings.laser.LaserEmbeddings)
+| `2204.00498v1` [Evaluating the Text-to-SQL Capabilities of Large Language Models](http://arxiv.org/abs/2204.00498v1) | Nitarshan Rajkumar, Raymond Li, Dzmitry Bahdanau | 2022-03-15 | `API:` [langchain_community...SparkSQL](https://api.python.langchain.com/en/latest/utilities/langchain_community.utilities.spark_sql.SparkSQL.html#langchain_community.utilities.spark_sql.SparkSQL), [langchain_community...SQLDatabase](https://api.python.langchain.com/en/latest/utilities/langchain_community.utilities.sql_database.SQLDatabase.html#langchain_community.utilities.sql_database.SQLDatabase)
+| `2202.00666v5` [Locally Typical Sampling](http://arxiv.org/abs/2202.00666v5) | Clara Meister, Tiago Pimentel, Gian Wiher,  et al. | 2022-02-01 | `API:` [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint)
 | `2103.00020v1` [Learning Transferable Visual Models From Natural Language Supervision](http://arxiv.org/abs/2103.00020v1) | Alec Radford, Jong Wook Kim, Chris Hallacy,  et al. | 2021-02-26 | `API:` [langchain_experimental.open_clip](https://api.python.langchain.com/en/latest/experimental_api_reference.html#module-langchain_experimental.open_clip)
-| `1909.05858v2` [CTRL: A Conditional Transformer Language Model for Controllable Generation](http://arxiv.org/abs/1909.05858v2) | Nitish Shirish Keskar, Bryan McCann, Lav R. Varshney,  et al. | 2019-09-11 | `API:` [langchain_community.llms...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_community.llms...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_huggingface.llms...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint)
+| `1909.05858v2` [CTRL: A Conditional Transformer Language Model for Controllable Generation](http://arxiv.org/abs/1909.05858v2) | Nitish Shirish Keskar, Bryan McCann, Lav R. Varshney,  et al. | 2019-09-11 | `API:` [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint)
 | `1908.10084v1` [Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks](http://arxiv.org/abs/1908.10084v1) | Nils Reimers, Iryna Gurevych | 2019-08-27 | `Docs:` [docs/integrations/text_embedding/sentence_transformers](https://python.langchain.com/docs/integrations/text_embedding/sentence_transformers)

 ## Self-Discover: Large Language Models Self-Compose Reasoning Structures
@@ -415,7 +418,7 @@ publicly available.
 - **URL:** http://arxiv.org/abs/2304.03442v2
 - **LangChain:**

-   - **Cookbook:** [multiagent_bidding](https://github.com/langchain-ai/langchain/blob/master/cookbook/multiagent_bidding.ipynb), [generative_agents_interactive_simulacra_of_human_behavior](https://github.com/langchain-ai/langchain/blob/master/cookbook/generative_agents_interactive_simulacra_of_human_behavior.ipynb)
+   - **Cookbook:** [generative_agents_interactive_simulacra_of_human_behavior](https://github.com/langchain-ai/langchain/blob/master/cookbook/generative_agents_interactive_simulacra_of_human_behavior.ipynb), [multiagent_bidding](https://github.com/langchain-ai/langchain/blob/master/cookbook/multiagent_bidding.ipynb)

 **Abstract:** Believable proxies of human behavior can empower interactive applications
 ranging from immersive environments to rehearsal spaces for interpersonal
@@ -537,7 +540,7 @@ more than 1/1,000th the compute of GPT-4.
 - **URL:** http://arxiv.org/abs/2301.10226v4
 - **LangChain:**

-   - **API Reference:** [langchain_community.llms...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_community.llms...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_huggingface.llms...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_community.llms...OCIModelDeploymentTGI](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.oci_data_science_model_deployment_endpoint.OCIModelDeploymentTGI.html#langchain_community.llms.oci_data_science_model_deployment_endpoint.OCIModelDeploymentTGI)
+   - **API Reference:** [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_community...OCIModelDeploymentTGI](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.oci_data_science_model_deployment_endpoint.OCIModelDeploymentTGI.html#langchain_community.llms.oci_data_science_model_deployment_endpoint.OCIModelDeploymentTGI), [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint)

 **Abstract:** Potential harms of large language models can be mitigated by watermarking
 model output, i.e., embedding signals into generated text that are invisible to
@@ -562,7 +565,7 @@ family, and discuss robustness and security.
 - **URL:** http://arxiv.org/abs/2212.10496v1
 - **LangChain:**

-   - **API Reference:** [langchain.chains...HypotheticalDocumentEmbedder](https://api.python.langchain.com/en/latest/chains/langchain.chains.hyde.base.HypotheticalDocumentEmbedder.html#langchain.chains.hyde.base.HypotheticalDocumentEmbedder)
+   - **API Reference:** [langchain...HypotheticalDocumentEmbedder](https://api.python.langchain.com/en/latest/chains/langchain.chains.hyde.base.HypotheticalDocumentEmbedder.html#langchain.chains.hyde.base.HypotheticalDocumentEmbedder)
   - **Template:** [hyde](https://python.langchain.com/docs/templates/hyde)
   - **Cookbook:** [hypothetical_document_embeddings](https://github.com/langchain-ai/langchain/blob/master/cookbook/hypothetical_document_embeddings.ipynb)

@@ -626,7 +629,7 @@ further work on logical fallacy identification.
 - **URL:** http://arxiv.org/abs/2211.13892v2
 - **LangChain:**

-   - **API Reference:** [langchain_core.example_selectors...MaxMarginalRelevanceExampleSelector](https://api.python.langchain.com/en/latest/example_selectors/langchain_core.example_selectors.semantic_similarity.MaxMarginalRelevanceExampleSelector.html#langchain_core.example_selectors.semantic_similarity.MaxMarginalRelevanceExampleSelector)
+   - **API Reference:** [langchain_core...MaxMarginalRelevanceExampleSelector](https://api.python.langchain.com/en/latest/example_selectors/langchain_core.example_selectors.semantic_similarity.MaxMarginalRelevanceExampleSelector.html#langchain_core.example_selectors.semantic_similarity.MaxMarginalRelevanceExampleSelector)

 **Abstract:** Large language models (LLMs) have exhibited remarkable capabilities in
 learning from explanations in prompts, but there has been limited understanding
@@ -654,7 +657,7 @@ performance across three real-world tasks on multiple LLMs.
 - **URL:** http://arxiv.org/abs/2211.10435v2
 - **LangChain:**

-   - **API Reference:** [langchain_experimental.pal_chain](https://api.python.langchain.com/en/latest/experimental_api_reference.html#module-langchain_experimental.pal_chain), [langchain_experimental.pal_chain...PALChain](https://api.python.langchain.com/en/latest/pal_chain/langchain_experimental.pal_chain.base.PALChain.html#langchain_experimental.pal_chain.base.PALChain)
+   - **API Reference:** [langchain_experimental...PALChain](https://api.python.langchain.com/en/latest/pal_chain/langchain_experimental.pal_chain.base.PALChain.html#langchain_experimental.pal_chain.base.PALChain), [langchain_experimental.pal_chain](https://api.python.langchain.com/en/latest/experimental_api_reference.html#module-langchain_experimental.pal_chain)
   - **Cookbook:** [program_aided_language_model](https://github.com/langchain-ai/langchain/blob/master/cookbook/program_aided_language_model.ipynb)

 **Abstract:** Large language models (LLMs) have recently demonstrated an impressive ability
@@ -717,7 +720,7 @@ TensorFlow, JAX, and integrate with numerous MLOps tools.
 - **URL:** http://arxiv.org/abs/2205.12654v1
 - **LangChain:**

-   - **API Reference:** [langchain_community.embeddings...LaserEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.laser.LaserEmbeddings.html#langchain_community.embeddings.laser.LaserEmbeddings)
+   - **API Reference:** [langchain_community...LaserEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.laser.LaserEmbeddings.html#langchain_community.embeddings.laser.LaserEmbeddings)

 **Abstract:** Scaling multilingual representation learning beyond the hundred most frequent
 languages is challenging, in particular to cover the long tail of low-resource
@@ -746,7 +749,7 @@ encoders, mine bitexts, and validate the bitexts by training NMT systems.
 - **URL:** http://arxiv.org/abs/2204.00498v1
 - **LangChain:**

-   - **API Reference:** [langchain_community.utilities...SparkSQL](https://api.python.langchain.com/en/latest/utilities/langchain_community.utilities.spark_sql.SparkSQL.html#langchain_community.utilities.spark_sql.SparkSQL), [langchain_community.utilities...SQLDatabase](https://api.python.langchain.com/en/latest/utilities/langchain_community.utilities.sql_database.SQLDatabase.html#langchain_community.utilities.sql_database.SQLDatabase)
+   - **API Reference:** [langchain_community...SparkSQL](https://api.python.langchain.com/en/latest/utilities/langchain_community.utilities.spark_sql.SparkSQL.html#langchain_community.utilities.spark_sql.SparkSQL), [langchain_community...SQLDatabase](https://api.python.langchain.com/en/latest/utilities/langchain_community.utilities.sql_database.SQLDatabase.html#langchain_community.utilities.sql_database.SQLDatabase)

 **Abstract:** We perform an empirical evaluation of Text-to-SQL capabilities of the Codex
 language model. We find that, without any finetuning, Codex is a strong
@@ -765,7 +768,7 @@ few-shot examples.
 - **URL:** http://arxiv.org/abs/2202.00666v5
 - **LangChain:**

-   - **API Reference:** [langchain_community.llms...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_community.llms...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_huggingface.llms...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint)
+   - **API Reference:** [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint)

 **Abstract:** Today's probabilistic language generators fall short when it comes to
 producing coherent and fluent text despite the fact that the underlying models
@@ -829,7 +832,7 @@ https://github.com/OpenAI/CLIP.
 - **URL:** http://arxiv.org/abs/1909.05858v2
 - **LangChain:**

-   - **API Reference:** [langchain_community.llms...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_community.llms...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_huggingface.llms...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint)
+   - **API Reference:** [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint)

 **Abstract:** Large-scale language models show promising text generation capabilities, but
 users cannot easily control particular aspects of the generated text. We
--- a/docs/docs/concepts.mdx
+++ b/docs/docs/concepts.mdx
@@ -1,7 +1,3 @@
---
-keywords: [prompt, documents, chatprompttemplate, prompttemplate, invoke, lcel, tool, tools, embedding, embeddings, vector, vectorstore, llm, loader, retriever, retrievers]
---
-
 # Conceptual guide

 import ThemedImage from '@theme/ThemedImage';
@@ -62,6 +58,7 @@ A developer platform that lets you debug, test, evaluate, and monitor LLM applic
 />

 ## LangChain Expression Language (LCEL)
+<span data-heading-keywords="lcel"></span>

 LangChain Expression Language, or LCEL, is a declarative way to chain LangChain components.
 LCEL was designed from day 1 to **support putting prototypes in production, with no code changes**, from the simplest “prompt + LLM” chain to the most complex chains (we’ve seen folks successfully run LCEL chains with 100s of steps in production). To highlight a few of the reasons you might want to use LCEL:
@@ -92,15 +89,16 @@ With LCEL, **all** steps are automatically logged to [LangSmith](https://docs.sm
 Any chain created with LCEL can be easily deployed using [LangServe](/docs/langserve).

 ### Runnable interface
+<span data-heading-keywords="invoke"></span>

 To make it as easy as possible to create custom chains, we've implemented a ["Runnable"](https://api.python.langchain.com/en/stable/runnables/langchain_core.runnables.base.Runnable.html#langchain_core.runnables.base.Runnable) protocol. Many LangChain components implement the `Runnable` protocol, including chat models, LLMs, output parsers, retrievers, prompt templates, and more. There are also several useful primitives for working with runnables, which you can read about below.

 This is a standard interface, which makes it easy to define custom chains as well as invoke them in a standard way.
 The standard interface includes:

- [`stream`](#stream): stream back chunks of the response
- [`invoke`](#invoke): call the chain on an input
- [`batch`](#batch): call the chain on a list of inputs
+- `stream`: stream back chunks of the response
+- `invoke`: call the chain on an input
+- `batch`: call the chain on a list of inputs

 These also have corresponding async methods that should be used with [asyncio](https://docs.python.org/3/library/asyncio.html) `await` syntax for concurrency:

@@ -132,16 +130,17 @@ LangChain provides standard, extendable interfaces and external integrations for
 Some components LangChain implements, some components we rely on third-party integrations for, and others are a mix.

 ### Chat models
+<span data-heading-keywords="chat model,chat models"></span>

 Language models that use a sequence of messages as inputs and return chat messages as outputs (as opposed to using plain text).
-These are traditionally newer models (older models are generally `LLMs`, see above).
+These are traditionally newer models (older models are generally `LLMs`, see below).
 Chat models support the assignment of distinct roles to conversation messages, helping to distinguish messages from the AI, users, and instructions such as system messages.

 Although the underlying models are messages in, message out, the LangChain wrappers also allow these models to take a string as input. This means you can easily use chat models in place of LLMs.

-When a string is passed in as input, it is converted to a HumanMessage and then passed to the underlying model.
+When a string is passed in as input, it is converted to a `HumanMessage` and then passed to the underlying model.

-LangChain does not provide any ChatModels, rather we rely on third party integrations.
+LangChain does not host any Chat Models, rather we rely on third party integrations.

 We have some standardized parameters when constructing ChatModels:
 - `model`: the name of the model
@@ -154,16 +153,31 @@ Generally, such models are better at tool calling than non-fine-tuned models, an
 Please see the [tool calling section](/docs/concepts/#functiontool-calling) for more information.
 :::

+For specifics on how to use chat models, see the [relevant how-to guides here](/docs/how_to/#chat-models).
+
+#### Multimodality
+
+Some chat models are multimodal, accepting images, audio and even video as inputs. These are still less common, meaning model providers haven't standardized on the "best" way to define the API. Multimodal **outputs** are even less common. As such, we've kept our multimodal abstractions fairly light weight and plan to further solidify the multimodal APIs and interaction patterns as the field matures.
+
+In LangChain, most chat models that support multimodal inputs also accept those values in OpenAI's content blocks format. So far this is restricted to image inputs. For models like Gemini which support video and other bytes input, the APIs also support the native, model-specific representations.
+
+For specifics on how to use multimodal models, see the [relevant how-to guides here](/docs/how_to/#multimodal).
+
+For a full list of LangChain model providers with multimodal models, [check out this table](/docs/integrations/chat/#advanced-features).
+
 ### LLMs
+<span data-heading-keywords="llm,llms"></span>

 Language models that takes a string as input and returns a string.
-These are traditionally older models (newer models generally are `ChatModels`, see below).
+These are traditionally older models (newer models generally are [Chat Models](/docs/concepts/#chat-models), see below).

 Although the underlying models are string in, string out, the LangChain wrappers also allow these models to take messages as input.
-This makes them interchangeable with ChatModels.
+This gives them the same interface as [Chat Models](/docs/concepts/#chat-models).
 When messages are passed in as input, they will be formatted into a string under the hood before being passed to the underlying model.

-LangChain does not provide any LLMs, rather we rely on third party integrations.
+LangChain does not host any LLMs, rather we rely on third party integrations.
+
+For specifics on how to use LLMs, see the [relevant how-to guides here](/docs/how_to/#llms).

 ### Messages

@@ -218,6 +232,8 @@ This represents the result of a tool call. This is distinct from a FunctionMessa


 ### Prompt templates
+<span data-heading-keywords="prompt,prompttemplate,chatprompttemplate"></span>
+
 Prompt templates help to translate user input and parameters into instructions for a language model.
 This can be used to guide a model's response, helping it understand the context and generate relevant and coherent language-based output.

@@ -226,7 +242,7 @@ Prompt Templates take as input a dictionary, where each key represents a variabl
 Prompt Templates output a PromptValue. This PromptValue can be passed to an LLM or a ChatModel, and can also be cast to a string or a list of messages.
 The reason this PromptValue exists is to make it easy to switch between strings and messages.

-There are a few different types of prompt templates
+There are a few different types of prompt templates:

 #### String PromptTemplates

@@ -262,6 +278,7 @@ The first is a system message, that has no variables to format.
 The second is a HumanMessage, and will be formatted by the `topic` variable the user passes in.

 #### MessagesPlaceholder
+<span data-heading-keywords="messagesplaceholder"></span>

 This prompt template is responsible for adding a list of messages in a particular place.
 In the above ChatPromptTemplate, we saw how we could format two messages, each one a string.
@@ -293,14 +310,18 @@ prompt_template = ChatPromptTemplate.from_messages([
 ])
 ```

+For specifics on how to use prompt templates, see the [relevant how-to guides here](/docs/how_to/#prompt-templates).
+
 ### Example selectors
 One common prompting technique for achieving better performance is to include examples as part of the prompt.
 This gives the language model concrete examples of how it should behave.
 Sometimes these examples are hardcoded into the prompt, but for more advanced situations it may be nice to dynamically select them.
 Example Selectors are classes responsible for selecting and then formatting examples into prompts.

+For specifics on how to use example selectors, see the [relevant how-to guides here](/docs/how_to/#example-selectors).

 ### Output parsers
+<span data-heading-keywords="output parser"></span>

 :::note

@@ -344,16 +365,19 @@ LangChain has lots of different types of output parsers. This is a list of outpu
 | [Datetime](https://api.python.langchain.com/en/latest/output_parsers/langchain.output_parsers.datetime.DatetimeOutputParser.html#langchain.output_parsers.datetime.DatetimeOutputParser)        |                    | ✅                             |           | `str` \| `Message`                 | `datetime.datetime`  | Parses response into a datetime string.                                                                                                                                                                                                                  |
 | [Structured](https://api.python.langchain.com/en/latest/output_parsers/langchain.output_parsers.structured.StructuredOutputParser.html#langchain.output_parsers.structured.StructuredOutputParser)      |                    | ✅                             |           | `str` \| `Message`                 | `Dict[str, str]`     | An output parser that returns structured information. It is less powerful than other output parsers since it only allows for fields to be strings. This can be useful when you are working with smaller LLMs.                                            |

+For specifics on how to use output parsers, see the [relevant how-to guides here](/docs/how_to/#output-parsers).
+
 ### Chat history
 Most LLM applications have a conversational interface.
 An essential component of a conversation is being able to refer to information introduced earlier in the conversation.
 At bare minimum, a conversational system should be able to access some window of past messages directly.

 The concept of `ChatHistory` refers to a class in LangChain which can be used to wrap an arbitrary chain.
-This `ChatHistory` will keep track of inputs and outputs of the underlying chain, and append them as messages to a message database
+This `ChatHistory` will keep track of inputs and outputs of the underlying chain, and append them as messages to a message database.
 Future interactions will then load those messages and pass them into the chain as part of the input.

 ### Documents
+<span data-heading-keywords="document,documents"></span>

 A Document object in LangChain contains information about some data. It has two attributes:

@@ -361,6 +385,7 @@ A Document object in LangChain contains information about some data. It has two
 - `metadata: dict`: Arbitrary metadata associated with this document. Can track the document id, file name, etc.

 ### Document loaders
+<span data-heading-keywords="document loader,document loaders"></span>

 These classes load Document objects. LangChain has hundreds of integrations with various data sources to load data from: Slack, Notion, Google Drive, etc.

@@ -376,6 +401,8 @@ loader = CSVLoader(
 data = loader.load()
 ```

+For specifics on how to use document loaders, see the [relevant how-to guides here](/docs/how_to/#document-loaders).
+
 ### Text splitters

 Once you've loaded documents, you'll often want to transform them to better suit your application. The simplest example is you may want to split a long document into smaller chunks that can fit into your model's context window. LangChain has a number of built-in document transformers that make it easy to split, combine, filter, and otherwise manipulate documents.
@@ -393,14 +420,22 @@ That means there are two different axes along which you can customize your text
 1. How the text is split
 2. How the chunk size is measured

+For specifics on how to use text splitters, see the [relevant how-to guides here](/docs/how_to/#text-splitters).
+
 ### Embedding models
+<span data-heading-keywords="embedding,embeddings"></span>
+
 The Embeddings class is a class designed for interfacing with text embedding models. There are lots of embedding model providers (OpenAI, Cohere, Hugging Face, etc) - this class is designed to provide a standard interface for all of them.

 Embeddings create a vector representation of a piece of text. This is useful because it means we can think about text in the vector space, and do things like semantic search where we look for pieces of text that are most similar in the vector space.

 The base Embeddings class in LangChain provides two methods: one for embedding documents and one for embedding a query. The former takes as input multiple texts, while the latter takes a single text. The reason for having these as two separate methods is that some embedding providers have different embedding methods for documents (to be searched over) vs queries (the search query itself).

+For specifics on how to use embedding models, see the [relevant how-to guides here](/docs/how_to/#embedding-models).
+
 ### Vector stores
+<span data-heading-keywords="vector,vectorstore,vectorstores,vector store,vector stores"></span>
+
 One of the most common ways to store and search over unstructured data is to embed it and store the resulting embedding vectors,
 and then at query time to embed the unstructured query and retrieve the embedding vectors that are 'most similar' to the embedded query.
 A vector store takes care of storing embedded data and performing vector search for you.
@@ -412,7 +447,11 @@ vectorstore = MyVectorStore()
 retriever = vectorstore.as_retriever()
 ```

+For specifics on how to use vector stores, see the [relevant how-to guides here](/docs/how_to/#vector-stores).
+
 ### Retrievers
+<span data-heading-keywords="retriever,retrievers"></span>
+
 A retriever is an interface that returns documents given an unstructured query.
 It is more general than a vector store.
 A retriever does not need to be able to store documents, only to return (or retrieve) them.
@@ -420,7 +459,10 @@ Retrievers can be created from vectorstores, but are also broad enough to includ

 Retrievers accept a string query as input and return a list of Document's as output.

+For specifics on how to use retrievers, see the [relevant how-to guides here](/docs/how_to/#retrievers).
+
 ### Tools
+<span data-heading-keywords="tool,tools"></span>

 Tools are interfaces that an agent, a chain, or a chat model / LLM can use to interact with the world.

@@ -446,6 +488,8 @@ Generally, when designing tools to be used by a chat model or LLM, it is importa
 - Models will perform better if the tools have well-chosen names, descriptions, and JSON schemas.
 - Simpler tools are generally easier for models to use than more complex tools.

+For specifics on how to use tools, see the [relevant how-to guides here](/docs/how_to/#tools).
+
 ### Toolkits

 Toolkits are collections of tools that are designed to be used together for specific tasks. They have convenient loading methods.
@@ -465,7 +509,7 @@ tools = toolkit.get_tools()

 By themselves, language models can't take actions - they just output text.
 A big use case for LangChain is creating **agents**.
-Agents are systems that use an LLM as a reasoning enginer to determine which actions to take and what the inputs to those actions should be.
+Agents are systems that use an LLM as a reasoning engine to determine which actions to take and what the inputs to those actions should be.
 The results of those actions can then be fed back into the agent and it determine whether more actions are needed, or whether it is okay to finish.

 [LangGraph](https://github.com/langchain-ai/langgraph) is an extension of LangChain specifically aimed at creating highly controllable and customizable agents.
@@ -478,13 +522,7 @@ In order to solve that we built LangGraph to be this flexible, highly-controllab

 If you are still using AgentExecutor, do not fear: we still have a guide on [how to use AgentExecutor](/docs/how_to/agent_executor).
 It is recommended, however, that you start to transition to LangGraph.
-In order to assist in this we have put together a [transition guide on how to do so](/docs/how_to/migrate_agent)
-
-### Multimodal
-
-Some models are multimodal, accepting images, audio and even video as inputs. These are still less common, meaning model providers haven't standardized on the "best" way to define the API. Multimodal **outputs** are even less common. As such, we've kept our multimodal abstractions fairly light weight and plan to further solidify the multimodal APIs and interaction patterns as the field matures.
-
-In LangChain, most chat models that support multimodal inputs also accept those values in OpenAI's content blocks format. So far this is restricted to image inputs. For models like Gemini which support video and other bytes input, the APIs also support the native, model-specific representations.
+In order to assist in this we have put together a [transition guide on how to do so](/docs/how_to/migrate_agent).

 ### Callbacks

@@ -556,9 +594,215 @@ This is a common reason why you may fail to see events being emitted from custom
 runnables or tools.
 :::

+For specifics on how to use callbacks, see the [relevant how-to guides here](/docs/how_to/#callbacks).
+
 ## Techniques

-### Function/tool calling
+### Streaming
+
+Individual LLM calls often run for much longer than traditional resource requests.
+This compounds when you build more complex chains or agents that require multiple reasoning steps.
+
+Fortunately, LLMs generate output iteratively, which means it's possible to show sensible intermediate results
+before the final response is ready. Consuming output as soon as it becomes available has therefore become a vital part of the UX
+around building apps with LLMs to help alleviate latency issues, and LangChain aims to have first-class support for streaming.
+
+Below, we'll discuss some concepts and considerations around streaming in LangChain.
+
+#### Tokens
+
+The unit that most model providers use to measure input and output is via a unit called a **token**.
+Tokens are the basic units that language models read and generate when processing or producing text.
+The exact definition of a token can vary depending on the specific way the model was trained -
+for instance, in English, a token could be a single word like "apple", or a part of a word like "app".
+
+When you send a model a prompt, the words and characters in the prompt are encoded into tokens using a **tokenizer**.
+The model then streams back generated output tokens, which the tokenizer decodes into human-readable text.
+The below example shows how OpenAI models tokenize `LangChain is cool!`:
+
+![](/img/tokenization.png)
+
+You can see that it gets split into 5 different tokens, and that the boundaries between tokens are not exactly the same as word boundaries.
+
+The reason language models use tokens rather than something more immediately intuitive like "characters"
+has to do with how they process and understand text. At a high-level, language models iteratively predict their next generated output based on
+the initial input and their previous generations. Training the model using tokens language models to handle linguistic
+units (like words or subwords) that carry meaning, rather than individual characters, which makes it easier for the model
+to learn and understand the structure of the language, including grammar and context.
+Furthermore, using tokens can also improve efficiency, since the model processes fewer units of text compared to character-level processing.
+
+#### Callbacks
+
+The lowest level way to stream outputs from LLMs in LangChain is via the [callbacks](/docs/concepts/#callbacks) system. You can pass a
+callback handler that handles the [`on_llm_new_token`](https://api.python.langchain.com/en/latest/callbacks/langchain.callbacks.streaming_aiter.AsyncIteratorCallbackHandler.html#langchain.callbacks.streaming_aiter.AsyncIteratorCallbackHandler.on_llm_new_token) event into LangChain components. When that component is invoked, any
+[LLM](/docs/concepts/#llms) or [chat model](/docs/concepts/#chat-models) contained in the component calls
+the callback with the generated token. Within the callback, you could pipe the tokens into some other destination, e.g. a HTTP response.
+You can also handle the [`on_llm_end`](https://api.python.langchain.com/en/latest/callbacks/langchain.callbacks.streaming_aiter.AsyncIteratorCallbackHandler.html#langchain.callbacks.streaming_aiter.AsyncIteratorCallbackHandler.on_llm_end) event to perform any necessary cleanup.
+
+You can see [this how-to section](/docs/how_to/#callbacks) for more specifics on using callbacks.
+
+Callbacks were the first technique for streaming introduced in LangChain. While powerful and generalizable,
+they can be unwieldy for developers. For example:
+
+- You need to explicitly initialize and manage some aggregator or other stream to collect results.
+- The execution order isn't explicitly guaranteed, and you could theoretically have a callback run after the `.invoke()` method finishes.
+- Providers would often make you pass an additional parameter to stream outputs instead of returning them all at once.
+- You would often ignore the result of the actual model call in favor of callback results.
+
+#### `.stream()` and `.astream()`
+
+LangChain also includes the `.stream()` method (and the equivalent `.astream()` method for [async](https://docs.python.org/3/library/asyncio.html) environments) as a more ergonomic streaming interface.
+`.stream()` returns an iterator, which you can consume with a simple `for` loop. Here's an example with a chat model:
+
+```python
+from langchain_anthropic import ChatAnthropic
+
+model = ChatAnthropic(model="claude-3-sonnet-20240229")
+
+for chunk in model.stream("what color is the sky?"):
+    print(chunk.content, end="|", flush=True)
+```
+
+For models (or other components) that don't support streaming natively, this iterator would just yield a single chunk, but
+you could still use the same general pattern. Using `.stream()` will also automatically call the model in streaming mode
+without the need to provide additional config.
+
+The type of each outputted chunk depends on the type of component - for example, chat models yield [`AIMessageChunks`](https://api.python.langchain.com/en/latest/messages/langchain_core.messages.ai.AIMessageChunk.html).
+Because this method is part of [LangChain Expression Language](/docs/concepts/#langchain-expression-language-lcel),
+you can handle formatting differences from different outputs using an [output parser](/docs/concepts/#output-parsers) to transform
+each yielded chunk.
+
+You can check out [this guide](/docs/how_to/streaming/#using-stream) for more detail on how to use `.stream()`.
+
+#### `.astream_events()`
+
+While the `.stream()` method is easier to use than callbacks, it only returns one type of value. This is fine for single LLM calls,
+but as you build more complex chains of several LLM calls together, you may want to use the intermediate values of
+the chain alongside the final output - for example, returning sources alongside the final generation when building a chat
+over documents app.
+
+There are ways to do this using the aforementioned callbacks, or by constructing your chain in such a way that it passes intermediate
+values to the end with something like [`.assign()`](/docs/how_to/passthrough/), but LangChain also includes an
+`.astream_events()` method that combines the flexibility of callbacks with the ergonomics of `.stream()`. When called, it returns an iterator
+which yields [various types of events](/docs/how_to/streaming/#event-reference) that you can filter and process according
+to the needs of your project.
+
+Here's one small example that prints just events containing streamed chat model output:
+
+```python
+from langchain_core.output_parsers import StrOutputParser
+from langchain_core.prompts import ChatPromptTemplate
+from langchain_anthropic import ChatAnthropic
+
+model = ChatAnthropic(model="claude-3-sonnet-20240229")
+
+prompt = ChatPromptTemplate.from_template("tell me a joke about {topic}")
+parser = StrOutputParser()
+chain = prompt | model | parser
+
+async for event in chain.astream_events({"topic": "parrot"}, version="v2"):
+    kind = event["event"]
+    if kind == "on_chat_model_stream":
+        print(event, end="|", flush=True)
+```
+
+You can roughly think of it as an iterator over callback events (though the format differs) - and you can use it on almost all LangChain components!
+
+See [this guide](/docs/how_to/streaming/#using-stream-events) for more detailed information on how to use `.astream_events()`.
+
+### Structured output
+
+LLMs are capable of generating arbitrary text. This enables the model to respond appropriately to a wide
+range of inputs, but for some use-cases, it can be useful to constrain the LLM's output
+to a specific format or structure. This is referred to as **structured output**.
+
+For example, if the output is to be stored in a relational database,
+it is much easier if the model generates output that adheres to a defined schema or format.
+[Extracting specific information](/docs/tutorials/extraction/) from unstructured text is another
+case where this is particularly useful. Most commonly, the output format will be JSON,
+though other formats such as [YAML](/docs/how_to/output_parser_yaml/) can be useful too. Below, we'll discuss
+a few ways to get structured output from models in LangChain.
+
+#### `.with_structured_output()`
+
+For convenience, some LangChain chat models support a `.with_structured_output()` method.
+This method only requires a schema as input, and returns a dict or Pydantic object.
+Generally, this method is only present on models that support one of the more advanced methods described below,
+and will use one of them under the hood. It takes care of importing a suitable output parser and
+formatting the schema in the right format for the model.
+
+For more information, check out this [how-to guide](/docs/how_to/structured_output/#the-with_structured_output-method).
+
+#### Raw prompting
+
+The most intuitive way to get a model to structure output is to ask nicely.
+In addition to your query, you can give instructions describing what kind of output you'd like, then
+parse the output using an [output parser](/docs/concepts/#output-parsers) to convert the raw
+model message or string output into something more easily manipulated.
+
+The biggest benefit to raw prompting is its flexibility:
+
+- Raw prompting does not require any special model features, only sufficient reasoning capability to understand
+the passed schema.
+- You can prompt for any format you'd like, not just JSON. This can be useful if the model you
+are using is more heavily trained on a certain type of data, such as XML or YAML.
+
+However, there are some drawbacks too:
+
+- LLMs are non-deterministic, and prompting a LLM to consistently output data in the exactly correct format
+for smooth parsing can be surprisingly difficult and model-specific.
+- Individual models have quirks depending on the data they were trained on, and optimizing prompts can be quite difficult.
+Some may be better at interpreting [JSON schema](https://json-schema.org/), others may be best with TypeScript definitions,
+and still others may prefer XML.
+
+While we'll next go over some ways that you can take advantage of features offered by
+model providers to increase reliability, prompting techniques remain important for tuning your
+results no matter what method you choose.
+
+#### JSON mode
+<span data-heading-keywords="json mode"></span>
+
+Some models, such as [Mistral](/docs/integrations/chat/mistralai/), [OpenAI](/docs/integrations/chat/openai/),
+[Together AI](/docs/integrations/chat/together/) and [Ollama](/docs/integrations/chat/ollama/),
+support a feature called **JSON mode**, usually enabled via config.
+
+When enabled, JSON mode will constrain the model's output to always be some sort of valid JSON.
+Often they require some custom prompting, but it's usually much less burdensome and along the lines of,
+`"you must always return JSON"`, and the [output is easier to parse](/docs/how_to/output_parser_json/).
+
+It's also generally simpler and more commonly available than tool calling.
+
+Here's an example:
+
+```python
+from langchain_core.prompts import ChatPromptTemplate
+from langchain_openai import ChatOpenAI
+from langchain.output_parsers.json import SimpleJsonOutputParser
+
+model = ChatOpenAI(
+    model="gpt-4o",
+    model_kwargs={ "response_format": { "type": "json_object" } },
+)
+
+prompt = ChatPromptTemplate.from_template(
+    "Answer the user's question to the best of your ability."
+    'You must always output a JSON object with an "answer" key and a "followup_question" key.'
+    "{question}"
+)
+
+chain = prompt | model | SimpleJsonOutputParser()
+
+chain.invoke({ "question": "What is the powerhouse of the cell?" })
+```
+
+```
+{'answer': 'The powerhouse of the cell is the mitochondrion. It is responsible for producing energy in the form of ATP through cellular respiration.',
+ 'followup_question': 'Would you like to know more about how mitochondria produce energy?'}
+```
+
+For a full list of model providers that support JSON mode, see [this table](/docs/integrations/chat/#advanced-features).
+
+#### Function/tool calling

 :::info
 We use the term tool calling interchangeably with function calling. Although
@@ -576,8 +820,10 @@ from unstructured text, you could give the model an "extraction" tool that takes
 parameters matching the desired schema, then treat the generated output as your final
 result.

-A tool call includes a name, arguments dict, and an optional identifier. The
-arguments dict is structured `{argument_name: argument_value}`.
+For models that support it, tool calling can be very convenient. It removes the
+guesswork around how best to prompt schemas in favor of a built-in model feature. It can also
+more naturally support agentic flows, since you can just pass multiple tool schemas instead
+of fiddling with enums or unions.

 Many LLM providers, including [Anthropic](https://www.anthropic.com/),
 [Cohere](https://cohere.com/), [Google](https://cloud.google.com/vertex-ai),
@@ -594,14 +840,16 @@ LangChain provides a standardized interface for tool calling that is consistent

 The standard interface consists of:

-* `ChatModel.bind_tools()`: a method for specifying which tools are available for a model to call.
+* `ChatModel.bind_tools()`: a method for specifying which tools are available for a model to call. This method accepts [LangChain tools](/docs/concepts/#tools) here.
 * `AIMessage.tool_calls`: an attribute on the `AIMessage` returned from the model for accessing the tool calls requested by the model.

-There are two main use cases for function/tool calling:
+The following how-to guides are good practical resources for using function/tool calling:

 - [How to return structured data from an LLM](/docs/how_to/structured_output/)
 - [How to use a model to call tools](/docs/how_to/tool_calling/)

+For a full list of model providers that support tool calling, [see this table](/docs/integrations/chat/#advanced-features).
+
 ### Retrieval

 LangChain provides several advanced retrieval types. A full list is below, along with the following information:
@@ -627,6 +875,7 @@ LangChain provides several advanced retrieval types. A full list is below, along
 | [Multi-Query Retriever](/docs/how_to/MultiQueryRetriever/)     | Any                          | Yes                       | If users are asking questions that are complex and require multiple pieces of distinct information to respond                                 | This uses an LLM to generate multiple queries from the original one. This is useful when the original query needs pieces of information about multiple topics to be properly answered. By generating multiple queries, we can then fetch documents for each of them.                             |
 | [Ensemble](/docs/how_to/ensemble_retriever/)                  | Any                          | No                        | If you have multiple retrieval methods and want to try combining them.                                                                        | This fetches documents from multiple retrievers and then combines them.                                                                                                                                                                                                                          |

+For a high-level guide on retrieval, see this [tutorial on RAG](/docs/tutorials/rag/).

 ### Text splitting

--- a/docs/docs/contributing/code.mdx
+++ b/docs/docs/contributing/code.mdx
@@ -206,9 +206,7 @@ ignore-words-list = 'momento,collison,ned,foor,reworkd,parth,whats,aapply,mysogy

 `langchain-core` and partner packages **do not use** optional dependencies in this way.

-You only need to add a new dependency if a **unit test** relies on the package.
-If your package is only required for **integration tests**, then you can skip these
-steps and leave all pyproject.toml and poetry.lock files alone.
+You'll notice that `pyproject.toml` and `poetry.lock` are **not** touched when you add optional dependencies below.

 If you're adding a new dependency to Langchain, assume that it will be an optional dependency, and
 that most users won't have it installed.
@@ -216,20 +214,12 @@ that most users won't have it installed.
 Users who do not have the dependency installed should be able to **import** your code without
 any side effects (no warnings, no errors, no exceptions).

-To introduce the dependency to the pyproject.toml file correctly, please do the following:
+To introduce the dependency to a library, please do the following:

-1. Add the dependency to the main group as an optional dependency
-  ```bash
-  poetry add --optional [package_name]
-  ```
-2. Open pyproject.toml and add the dependency to the `extended_testing` extra
-3. Relock the poetry file to update the extra.
-  ```bash
-  poetry lock --no-update
-  ```
-4. Add a unit test that the very least attempts to import the new code. Ideally, the unit
+1. Open extended_testing_deps.txt and add the dependency
+2. Add a unit test that the very least attempts to import the new code. Ideally, the unit
 test makes use of lightweight fixtures to test the logic of the code.
-5. Please use the `@pytest.mark.requires(package_name)` decorator for any tests that require the dependency.
+3. Please use the `@pytest.mark.requires(package_name)` decorator for any unit tests that require the dependency.

 ## Adding a Jupyter Notebook

--- a/docs/docs/contributing/documentation/style_guide.mdx
+++ b/docs/docs/contributing/documentation/style_guide.mdx
@@ -55,7 +55,7 @@ The below sections are listed roughly in order of increasing level of abstractio

 ### Expression Language

-[LangChain Expression Language (LCEL)](/docs/concepts#langchain-expression-language) is the fundamental way that most LangChain components fit together, and this section is designed to teach
+[LangChain Expression Language (LCEL)](/docs/concepts#langchain-expression-language-lcel) is the fundamental way that most LangChain components fit together, and this section is designed to teach
 developers how to use it to build with LangChain's primitives effectively.

 This section should contains **Tutorials** that teach how to stream and use LCEL primitives for more abstract tasks, **Explanations** of specific behaviors,
--- a/docs/docs/contributing/index.mdx
+++ b/docs/docs/contributing/index.mdx
@@ -48,7 +48,7 @@ In a similar vein, we do enforce certain linting, formatting, and documentation
 If you are finding these difficult (or even just annoying) to work with, feel free to contact a maintainer for help -
 we do not want these to get in the way of getting good code into the codebase.

-# 🌟 Recognition
+### 🌟 Recognition

 If your contribution has made its way into a release, we will want to give you credit on Twitter (only if you want though)!
-If you have a Twitter account you would like us to mention, please let us know in the PR or through another means.
+If you have a Twitter account you would like us to mention, please let us know in the PR or through another means.
--- a/docs/docs/contributing/repo_structure.mdx
+++ b/docs/docs/contributing/repo_structure.mdx
@@ -15,12 +15,22 @@ Here's the structure visualized as a tree:
 ├── cookbook # Tutorials and examples
 ├── docs # Contains content for the documentation here: https://python.langchain.com/
 ├── libs
-│   ├── langchain # Main package
+│   ├── langchain
+│   │   ├── langchain
 │   │   ├── tests/unit_tests # Unit tests (present in each package not shown for brevity)
 │   │   ├── tests/integration_tests # Integration tests (present in each package not shown for brevity)
-│   ├── langchain-community # Third-party integrations
-│   ├── langchain-core # Base interfaces for key abstractions
-│   ├── langchain-experimental # Experimental components and chains
+│   ├── community # Third-party integrations
+│   │   ├── langchain-community
+│   ├── core # Base interfaces for key abstractions
+│   │   ├── langchain-core
+│   ├── experimental # Experimental components and chains
+│   │   ├── langchain-experimental
+|   ├── cli # Command line interface
+│   │   ├── langchain-cli
+│   ├── text-splitters
+│   │   ├── langchain-text-splitters
+│   ├── standard-tests
+│   │   ├── langchain-standard-tests
 │   ├── partners
 │       ├── langchain-partner-1
 │       ├── langchain-partner-2
--- a/docs/docs/example_data/nke-10k-2023.pdf
+++ b/docs/docs/example_data/nke-10k-2023.pdf
--- a/docs/docs/how_to/agent_executor.ipynb
+++ b/docs/docs/how_to/agent_executor.ipynb
@@ -15,7 +15,11 @@
   "id": "f4c03f40-1328-412d-8a48-1db0cd481b77",
   "metadata": {},
   "source": [
-    "# Build an Agent\n",
+    "# Build an Agent with AgentExecutor (Legacy)\n",
+    "\n",
+    ":::{.callout-important}\n",
+    "This section will cover building with the legacy LangChain AgentExecutor. These are fine for getting started, but past a certain point, you will likely want flexibility and control that they do not offer. For working with more advanced agents, we'd recommend checking out [LangGraph Agents](/docs/concepts/#langgraph) or the [migration guide](/docs/how_to/migrate_agent/)\n",
+    ":::\n",
    "\n",
    "By themselves, language models can't take actions - they just output text.\n",
    "A big use case for LangChain is creating **agents**.\n",
@@ -24,10 +28,6 @@
    "\n",
    "In this tutorial, we will build an agent that can interact with multiple different tools: one being a local database, the other being a search engine. You will be able to ask this agent questions, watch it call tools, and have conversations with it.\n",
    "\n",
-    ":::{.callout-important}\n",
-    "This section will cover building with LangChain Agents. LangChain Agents are fine for getting started, but past a certain point, you will likely want flexibility and control that they do not offer. For working with more advanced agents, we'd reccommend checking out [LangGraph](/docs/concepts/#langgraph)\n",
-    ":::\n",
-    "\n",
    "## Concepts\n",
    "\n",
    "Concepts we will cover are:\n",
--- a/docs/docs/how_to/chat_models_universal_init.ipynb
+++ b/docs/docs/how_to/chat_models_universal_init.ipynb
@@ -0,0 +1,157 @@
+{
+ "cells": [
+  {
+   "cell_type": "markdown",
+   "id": "cfdf4f09-8125-4ed1-8063-6feed57da8a3",
+   "metadata": {},
+   "source": [
+    "# How to let your end users choose their model\n",
+    "\n",
+    "Many LLM applications let end users specify what model provider and model they want the application to be powered by. This requires writing some logic to initialize different ChatModels based on some user configuration. The `init_chat_model()` helper method makes it easy to initialize a number of different model integrations without having to worry about import paths and class names.\n",
+    "\n",
+    ":::tip Supported models\n",
+    "\n",
+    "See the [init_chat_model()](https://api.python.langchain.com/en/latest/chat_models/langchain.chat_models.base.init_chat_model.html) API reference for a full list of supported integrations.\n",
+    "\n",
+    "Make sure you have the integration packages installed for any model providers you want to support. E.g. you should have `langchain-openai` installed to init an OpenAI model.\n",
+    "\n",
+    ":::"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "165b0de6-9ae3-4e3d-aa98-4fc8a97c4a06",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "%pip install -qU langchain langchain-openai langchain-anthropic langchain-google-vertexai"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "ea2c9f57-a796-45f8-b6f4-3efd3f361a9b",
+   "metadata": {},
+   "source": [
+    "## Basic usage"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 5,
+   "id": "79e14913-803c-4382-9009-5c6af3d75d35",
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "GPT-4o: I'm an AI created by OpenAI, and I don't have a personal name. You can call me Assistant! How can I help you today?\n",
+      "\n",
+      "Claude Opus: My name is Claude. It's nice to meet you!\n",
+      "\n",
+      "Gemini 1.5: I am a large language model, trained by Google. I do not have a name. \n",
+      "\n",
+      "\n"
+     ]
+    }
+   ],
+   "source": [
+    "from langchain.chat_models import init_chat_model\n",
+    "\n",
+    "# Returns a langchain_openai.ChatOpenAI instance.\n",
+    "gpt_4o = init_chat_model(\"gpt-4o\", model_provider=\"openai\", temperature=0)\n",
+    "# Returns a langchain_anthropic.ChatAnthropic instance.\n",
+    "claude_opus = init_chat_model(\n",
+    "    \"claude-3-opus-20240229\", model_provider=\"anthropic\", temperature=0\n",
+    ")\n",
+    "# Returns a langchain_google_vertexai.ChatVertexAI instance.\n",
+    "gemini_15 = init_chat_model(\n",
+    "    \"gemini-1.5-pro\", model_provider=\"google_vertexai\", temperature=0\n",
+    ")\n",
+    "\n",
+    "# Since all model integrations implement the ChatModel interface, you can use them in the same way.\n",
+    "print(\"GPT-4o: \" + gpt_4o.invoke(\"what's your name\").content + \"\\n\")\n",
+    "print(\"Claude Opus: \" + claude_opus.invoke(\"what's your name\").content + \"\\n\")\n",
+    "print(\"Gemini 1.5: \" + gemini_15.invoke(\"what's your name\").content + \"\\n\")"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "fff9a4c8-b6ee-4a1a-8d3d-0ecaa312d4ed",
+   "metadata": {},
+   "source": [
+    "## Simple config example"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "75c25d39-bf47-4b51-a6c6-64d9c572bfd6",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "user_config = {\n",
+    "    \"model\": \"...user-specified...\",\n",
+    "    \"model_provider\": \"...user-specified...\",\n",
+    "    \"temperature\": 0,\n",
+    "    \"max_tokens\": 1000,\n",
+    "}\n",
+    "\n",
+    "llm = init_chat_model(**user_config)\n",
+    "llm.invoke(\"what's your name\")"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "f811f219-5e78-4b62-b495-915d52a22532",
+   "metadata": {},
+   "source": [
+    "## Inferring model provider\n",
+    "\n",
+    "For common and distinct model names `init_chat_model()` will attempt to infer the model provider. See the [API reference](https://api.python.langchain.com/en/latest/chat_models/langchain.chat_models.base.init_chat_model.html) for a full list of inference behavior. E.g. any model that starts with `gpt-3...` or `gpt-4...` will be inferred as using model provider `openai`."
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 4,
+   "id": "0378ccc6-95bc-4d50-be50-fccc193f0a71",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "gpt_4o = init_chat_model(\"gpt-4o\", temperature=0)\n",
+    "claude_opus = init_chat_model(\"claude-3-opus-20240229\", temperature=0)\n",
+    "gemini_15 = init_chat_model(\"gemini-1.5-pro\", temperature=0)"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "da07b5c0-d2e6-42e4-bfcd-2efcfaae6221",
+   "metadata": {},
+   "outputs": [],
+   "source": []
+  }
+ ],
+ "metadata": {
+  "kernelspec": {
+   "display_name": "poetry-venv-2",
+   "language": "python",
+   "name": "poetry-venv-2"
+  },
+  "language_info": {
+   "codemirror_mode": {
+    "name": "ipython",
+    "version": 3
+   },
+   "file_extension": ".py",
+   "mimetype": "text/x-python",
+   "name": "python",
+   "nbconvert_exporter": "python",
+   "pygments_lexer": "ipython3",
+   "version": "3.9.1"
+  }
+ },
+ "nbformat": 4,
+ "nbformat_minor": 5
+}
--- a/docs/docs/how_to/example_selectors.ipynb
+++ b/docs/docs/how_to/example_selectors.ipynb
@@ -60,7 +60,7 @@
   "source": [
    "examples = [\n",
    "    {\"input\": \"hi\", \"output\": \"ciao\"},\n",
-    "    {\"input\": \"bye\", \"output\": \"arrivaderci\"},\n",
+    "    {\"input\": \"bye\", \"output\": \"arrivederci\"},\n",
    "    {\"input\": \"soccer\", \"output\": \"calcio\"},\n",
    "]"
   ]
@@ -133,7 +133,7 @@
    {
     "data": {
      "text/plain": [
-       "[{'input': 'bye', 'output': 'arrivaderci'}]"
+       "[{'input': 'bye', 'output': 'arrivederci'}]"
      ]
     },
     "execution_count": 39,
@@ -209,7 +209,7 @@
     "name": "stdout",
     "output_type": "stream",
     "text": [
-      "Translate the following words from English to Italain:\n",
+      "Translate the following words from English to Italian:\n",
      "\n",
      "Input: hand -> Output: mano\n",
      "\n",
@@ -222,7 +222,7 @@
    "    example_selector=example_selector,\n",
    "    example_prompt=example_prompt,\n",
    "    suffix=\"Input: {input} -> Output:\",\n",
-    "    prefix=\"Translate the following words from English to Italain:\",\n",
+    "    prefix=\"Translate the following words from English to Italian:\",\n",
    "    input_variables=[\"input\"],\n",
    ")\n",
    "\n",
--- a/docs/docs/how_to/extraction_examples.ipynb
+++ b/docs/docs/how_to/extraction_examples.ipynb
@@ -128,7 +128,7 @@
    "    # Having a good description can help improve extraction results.\n",
    "    name: Optional[str] = Field(..., description=\"The name of the person\")\n",
    "    hair_color: Optional[str] = Field(\n",
-    "        ..., description=\"The color of the peron's eyes if known\"\n",
+    "        ..., description=\"The color of the person's hair if known\"\n",
    "    )\n",
    "    height_in_meters: Optional[str] = Field(..., description=\"Height in METERs\")\n",
    "\n",
--- a/docs/docs/how_to/index.mdx
+++ b/docs/docs/how_to/index.mdx
@@ -14,6 +14,7 @@ For comprehensive descriptions of every class and function see the [API Referenc
 ## Installation

 - [How to: install LangChain packages](/docs/how_to/installation/)
+- [How to: use LangChain with different Pydantic versions](/docs/how_to/pydantic_compatibility)

 ## Key features

@@ -49,7 +50,7 @@ These are the core building blocks you can use when building applications.

 ### Prompt templates

-Prompt Templates are responsible for formatting user input into a format that can be passed to a language model.
+[Prompt Templates](/docs/concepts/#prompt-templates) are responsible for formatting user input into a format that can be passed to a language model.

 - [How to: use few shot examples](/docs/how_to/few_shot_examples)
 - [How to: use few shot examples in chat models](/docs/how_to/few_shot_examples_chat/)
@@ -58,7 +59,7 @@ Prompt Templates are responsible for formatting user input into a format that ca

 ### Example selectors

-Example Selectors are responsible for selecting the correct few shot examples to pass to the prompt.
+[Example Selectors](/docs/concepts/#example-selectors) are responsible for selecting the correct few shot examples to pass to the prompt.

 - [How to: use example selectors](/docs/how_to/example_selectors)
 - [How to: select examples by length](/docs/how_to/example_selectors_length_based)
@@ -68,7 +69,7 @@ Example Selectors are responsible for selecting the correct few shot examples to

 ### Chat models

-Chat Models are newer forms of language models that take messages in and output a message.
+[Chat Models](/docs/concepts/#chat-models) are newer forms of language models that take messages in and output a message.

 - [How to: do function/tool calling](/docs/how_to/tool_calling)
 - [How to: get models to return structured output](/docs/how_to/structured_output)
@@ -78,10 +79,11 @@ Chat Models are newer forms of language models that take messages in and output
 - [How to: stream a response back](/docs/how_to/chat_streaming)
 - [How to: track token usage](/docs/how_to/chat_token_usage_tracking)
 - [How to: track response metadata across providers](/docs/how_to/response_metadata)
+- [How to: let your end users choose their model](/docs/how_to/chat_models_universal_init/)

 ### LLMs

-What LangChain calls LLMs are older forms of language models that take a string in and output a string.
+What LangChain calls [LLMs](/docs/concepts/#llms) are older forms of language models that take a string in and output a string.

 - [How to: cache model responses](/docs/how_to/llm_caching)
 - [How to: create a custom LLM class](/docs/how_to/custom_llm)
@@ -91,7 +93,7 @@ What LangChain calls LLMs are older forms of language models that take a string

 ### Output parsers

-Output Parsers are responsible for taking the output of an LLM and parsing into more structured format.
+[Output Parsers](/docs/concepts/#output-parsers) are responsible for taking the output of an LLM and parsing into more structured format.

 - [How to: use output parsers to parse an LLM response into structured format](/docs/how_to/output_parser_structured)
 - [How to: parse JSON output](/docs/how_to/output_parser_json)
@@ -103,7 +105,7 @@ Output Parsers are responsible for taking the output of an LLM and parsing into

 ### Document loaders

-Document Loaders are responsible for loading documents from a variety of sources.
+[Document Loaders](/docs/concepts/#document-loaders) are responsible for loading documents from a variety of sources.

 - [How to: load CSV data](/docs/how_to/document_loader_csv)
 - [How to: load data from a directory](/docs/how_to/document_loader_directory)
@@ -116,7 +118,7 @@ Document Loaders are responsible for loading documents from a variety of sources

 ### Text splitters

-Text Splitters take a document and split into chunks that can be used for retrieval.
+[Text Splitters](/docs/concepts/#text-splitters) take a document and split into chunks that can be used for retrieval.

 - [How to: recursively split text](/docs/how_to/recursive_text_splitter)
 - [How to: split by HTML headers](/docs/how_to/HTML_header_metadata_splitter)
@@ -130,20 +132,20 @@ Text Splitters take a document and split into chunks that can be used for retrie

 ### Embedding models

-Embedding Models take a piece of text and create a numerical representation of it.
+[Embedding Models](/docs/concepts/#embedding-models) take a piece of text and create a numerical representation of it.

 - [How to: embed text data](/docs/how_to/embed_text)
 - [How to: cache embedding results](/docs/how_to/caching_embeddings)

 ### Vector stores

-Vector stores are databases that can efficiently store and retrieve embeddings.
+[Vector stores](/docs/concepts/#vector-stores) are databases that can efficiently store and retrieve embeddings.

 - [How to: use a vector store to retrieve data](/docs/how_to/vectorstores)

 ### Retrievers

-Retrievers are responsible for taking a query and returning relevant documents.
+[Retrievers](/docs/concepts/#retrievers) are responsible for taking a query and returning relevant documents.

 - [How to: use a vector store to retrieve data](/docs/how_to/vectorstore_retriever)
 - [How to: generate multiple queries to retrieve data for](/docs/how_to/MultiQueryRetriever)
@@ -166,12 +168,13 @@ Indexing is the process of keeping your vectorstore in-sync with the underlying

 ### Tools

-LangChain Tools contain a description of the tool (to pass to the language model) as well as the implementation of the function to call).
+LangChain [Tools](/docs/concepts/#tools) contain a description of the tool (to pass to the language model) as well as the implementation of the function to call).

 - [How to: create custom tools](/docs/how_to/custom_tools)
 - [How to: use built-in tools and built-in toolkits](/docs/how_to/tools_builtin)
 - [How to: use a chat model to call tools](/docs/how_to/tool_calling/)
 - [How to: add ad-hoc tool calling capability to LLMs and chat models](/docs/how_to/tools_prompting)
+- [How to: pass run time values to tools](/docs/how_to/tool_runtime)
 - [How to: add a human in the loop to tool usage](/docs/how_to/tools_human)
 - [How to: handle errors when calling tools](/docs/how_to/tools_error)

@@ -194,6 +197,8 @@ For in depth how-to guides for agents, please check out [LangGraph](https://gith

 ### Callbacks

+[Callbacks](/docs/concepts/#callbacks) allow you to hook into the various stages of your LLM application's execution.
+
 - [How to: pass in callbacks at runtime](/docs/how_to/callbacks_runtime)
 - [How to: attach callbacks to a module](/docs/how_to/callbacks_attach)
 - [How to: pass callbacks into a module constructor](/docs/how_to/callbacks_constructor)
@@ -220,6 +225,7 @@ These guides cover use-case specific details.
 ### Q&A with RAG

 Retrieval Augmented Generation (RAG) is a way to connect LLMs to external sources of data.
+For a high-level tutorial on RAG, check out [this guide](/docs/tutorials/rag/).

 - [How to: add chat history](/docs/how_to/qa_chat_history_how_to/)
 - [How to: stream](/docs/how_to/qa_streaming/)
@@ -231,6 +237,7 @@ Retrieval Augmented Generation (RAG) is a way to connect LLMs to external source
 ### Extraction

 Extraction is when you use LLMs to extract structured information from unstructured text.
+For a high level tutorial on extraction, check out [this guide](/docs/tutorials/extraction/).

 - [How to: use reference examples](/docs/how_to/extraction_examples/)
 - [How to: handle long text](/docs/how_to/extraction_long_text/)
@@ -239,6 +246,7 @@ Extraction is when you use LLMs to extract structured information from unstructu
 ### Chatbots

 Chatbots involve using an LLM to have a conversation.
+For a high-level tutorial on building chatbots, check out [this guide](/docs/tutorials/chatbot/).

 - [How to: manage memory](/docs/how_to/chatbots_memory)
 - [How to: do retrieval](/docs/how_to/chatbots_retrieval)
@@ -247,6 +255,7 @@ Chatbots involve using an LLM to have a conversation.
 ### Query analysis

 Query Analysis is the task of using an LLM to generate a query to send to a retriever.
+For a high-level tutorial on query analysis, check out [this guide](/docs/tutorials/query_analysis/).

 - [How to: add examples to the prompt](/docs/how_to/query_few_shot)
 - [How to: handle cases where no queries are generated](/docs/how_to/query_no_queries)
@@ -258,6 +267,7 @@ Query Analysis is the task of using an LLM to generate a query to send to a retr
 ### Q&A over SQL + CSV

 You can use LLMs to do question answering over tabular data.
+For a high-level tutorial, check out [this guide](/docs/tutorials/sql_qa/).

 - [How to: use prompting to improve results](/docs/how_to/sql_prompting)
 - [How to: do query validation](/docs/how_to/sql_query_checking)
@@ -267,8 +277,25 @@ You can use LLMs to do question answering over tabular data.
 ### Q&A over graph databases

 You can use an LLM to do question answering over graph databases.
+For a high-level tutorial, check out [this guide](/docs/tutorials/graph/).

 - [How to: map values to a database](/docs/how_to/graph_mapping)
 - [How to: add a semantic layer over the database](/docs/how_to/graph_semantic)
 - [How to: improve results with prompting](/docs/how_to/graph_prompting)
 - [How to: construct knowledge graphs](/docs/how_to/graph_constructing)
+
+## [LangGraph](https://langchain-ai.github.io/langgraph)
+
+LangGraph is an extension of LangChain aimed at
+building robust and stateful multi-actor applications with LLMs by modeling steps as edges and nodes in a graph.
+
+LangGraph documentation is currently hosted on a separate site.
+You can peruse [LangGraph how-to guides here](https://langchain-ai.github.io/langgraph/how-tos/).
+
+## [LangSmith](https://docs.smith.langchain.com/)
+
+LangSmith allows you to closely trace, monitor and evaluate your LLM application.
+It seamlessly integrates with LangChain, and you can use it to inspect and debug individual steps of your chains as you build.
+
+LangSmith documentation is hosted on a separate site.
+You can peruse [LangSmith how-to guides here](https://docs.smith.langchain.com/how_to_guides/).
--- a/docs/docs/how_to/indexing.ipynb
+++ b/docs/docs/how_to/indexing.ipynb
@@ -60,7 +60,7 @@
    "   * document addition by id (`add_documents` method with `ids` argument)\n",
    "   * delete by id (`delete` method with `ids` argument)\n",
    "\n",
-    "Compatible Vectorstores: `Aerospike`, `AnalyticDB`, `AstraDB`, `AwaDB`, `Bagel`, `Cassandra`, `Chroma`, `CouchbaseVectorStore`, `DashVector`, `DatabricksVectorSearch`, `DeepLake`, `Dingo`, `ElasticVectorSearch`, `ElasticsearchStore`, `FAISS`, `HanaDB`, `Milvus`, `MyScale`, `OpenSearchVectorSearch`, `PGVector`, `Pinecone`, `Qdrant`, `Redis`, `Rockset`, `ScaNN`, `SupabaseVectorStore`, `SurrealDBStore`, `TimescaleVector`, `Vald`, `VDMS`, `Vearch`, `VespaStore`, `Weaviate`, `Yellowbrick`, `ZepVectorStore`, `TencentVectorDB`, `OpenSearchVectorSearch`.\n",
+    "Compatible Vectorstores: `Aerospike`, `AnalyticDB`, `AstraDB`, `AwaDB`, `AzureCosmosDBNoSqlVectorSearch`, `AzureCosmosDBVectorSearch`, `Bagel`, `Cassandra`, `Chroma`, `CouchbaseVectorStore`, `DashVector`, `DatabricksVectorSearch`, `DeepLake`, `Dingo`, `ElasticVectorSearch`, `ElasticsearchStore`, `FAISS`, `HanaDB`, `Milvus`, `MyScale`, `OpenSearchVectorSearch`, `PGVector`, `Pinecone`, `Qdrant`, `Redis`, `Rockset`, `ScaNN`, `SupabaseVectorStore`, `SurrealDBStore`, `TimescaleVector`, `Vald`, `VDMS`, `Vearch`, `VespaStore`, `Weaviate`, `Yellowbrick`, `ZepVectorStore`, `TencentVectorDB`, `OpenSearchVectorSearch`.\n",
    "  \n",
    "## Caution\n",
    "\n",
--- a/docs/docs/how_to/output_parser_structured.ipynb
+++ b/docs/docs/how_to/output_parser_structured.ipynb
@@ -94,7 +94,7 @@
   "source": [
    "## LCEL\n",
    "\n",
-    "Output parsers implement the [Runnable interface](/docs/concepts#interface), the basic building block of the [LangChain Expression Language (LCEL)](/docs/concepts#langchain-expression-language). This means they support `invoke`, `ainvoke`, `stream`, `astream`, `batch`, `abatch`, `astream_log` calls.\n",
+    "Output parsers implement the [Runnable interface](/docs/concepts#interface), the basic building block of the [LangChain Expression Language (LCEL)](/docs/concepts#langchain-expression-language-lcel). This means they support `invoke`, `ainvoke`, `stream`, `astream`, `batch`, `abatch`, `astream_log` calls.\n",
    "\n",
    "Output parsers accept a string or `BaseMessage` as input and can return an arbitrary type."
   ]
--- a/docs/docs/how_to/pydantic_compatibility.md
+++ b/docs/docs/how_to/pydantic_compatibility.md
@@ -0,0 +1,105 @@
+# How to use LangChain with different Pydantic versions
+
+- Pydantic v2 was released in June, 2023 (https://docs.pydantic.dev/2.0/blog/pydantic-v2-final/)
+- v2 contains has a number of breaking changes (https://docs.pydantic.dev/2.0/migration/)
+- Pydantic v2 and v1 are under the same package name, so both versions cannot be installed at the same time
+
+## LangChain Pydantic migration plan
+
+As of `langchain>=0.0.267`, LangChain will allow users to install either Pydantic V1 or V2. 
+   * Internally LangChain will continue to [use V1](https://docs.pydantic.dev/latest/migration/#continue-using-pydantic-v1-features).
+   * During this time, users can pin their pydantic version to v1 to avoid breaking changes, or start a partial
+   migration using pydantic v2 throughout their code, but avoiding mixing v1 and v2 code for LangChain (see below).
+
+User can either pin to pydantic v1, and upgrade their code in one go once LangChain has migrated to v2 internally, or they can start a partial migration to v2, but must avoid mixing v1 and v2 code for LangChain.
+
+Below are two examples of showing how to avoid mixing pydantic v1 and v2 code in
+the case of inheritance and in the case of passing objects to LangChain.
+
+**Example 1: Extending via inheritance**
+
+**YES** 
+
+```python
+from pydantic.v1 import root_validator, validator
+
+class CustomTool(BaseTool): # BaseTool is v1 code
+    x: int = Field(default=1)
+
+    def _run(*args, **kwargs):
+        return "hello"
+
+    @validator('x') # v1 code
+    @classmethod
+    def validate_x(cls, x: int) -> int:
+        return 1
+    
+
+CustomTool(
+    name='custom_tool',
+    description="hello",
+    x=1,
+)
+```
+
+Mixing Pydantic v2 primitives with Pydantic v1 primitives can raise cryptic errors
+
+**NO** 
+
+```python
+from pydantic import Field, field_validator # pydantic v2
+
+class CustomTool(BaseTool): # BaseTool is v1 code
+    x: int = Field(default=1)
+
+    def _run(*args, **kwargs):
+        return "hello"
+
+    @field_validator('x') # v2 code
+    @classmethod
+    def validate_x(cls, x: int) -> int:
+        return 1
+    
+
+CustomTool( 
+    name='custom_tool',
+    description="hello",
+    x=1,
+)
+```
+
+**Example 2: Passing objects to LangChain**
+
+**YES**
+
+```python
+from langchain_core.tools import Tool
+from pydantic.v1 import BaseModel, Field # <-- Uses v1 namespace
+
+class CalculatorInput(BaseModel):
+    question: str = Field()
+
+Tool.from_function( # <-- tool uses v1 namespace
+    func=lambda question: 'hello',
+    name="Calculator",
+    description="useful for when you need to answer questions about math",
+    args_schema=CalculatorInput
+)
+```
+
+**NO**
+
+```python
+from langchain_core.tools import Tool
+from pydantic import BaseModel, Field # <-- Uses v2 namespace
+
+class CalculatorInput(BaseModel):
+    question: str = Field()
+
+Tool.from_function( # <-- tool uses v1 namespace
+    func=lambda question: 'hello',
+    name="Calculator",
+    description="useful for when you need to answer questions about math",
+    args_schema=CalculatorInput
+)
+```
--- a/docs/docs/how_to/qa_sources.ipynb
+++ b/docs/docs/how_to/qa_sources.ipynb
@@ -14,7 +14,7 @@
    "We will cover two approaches:\n",
    "\n",
    "1. Using the built-in [create_retrieval_chain](https://api.python.langchain.com/en/latest/chains/langchain.chains.retrieval.create_retrieval_chain.html), which returns sources by default;\n",
-    "2. Using a simple [LCEL](/docs/concepts#langchain-expression-language) implementation, to show the operating principle."
+    "2. Using a simple [LCEL](/docs/concepts#langchain-expression-language-lcel) implementation, to show the operating principle."
   ]
  },
  {
--- a/docs/docs/how_to/sequence.ipynb
+++ b/docs/docs/how_to/sequence.ipynb
@@ -9,7 +9,7 @@
   },
   "source": [
    "---\n",
-    "keywords: [Runnable, Runnables, LCEL, chain, chains, chaining]\n",
+    "keywords: [Runnable, Runnables, RunnableSequence, LCEL, chain, chains, chaining]\n",
    "---"
   ]
  },
--- a/docs/docs/how_to/structured_output.ipynb
+++ b/docs/docs/how_to/structured_output.ipynb
@@ -3,10 +3,15 @@
  {
   "cell_type": "raw",
   "id": "27598444",
-   "metadata": {},
+   "metadata": {
+    "vscode": {
+     "languageId": "raw"
+    }
+   },
   "source": [
    "---\n",
    "sidebar_position: 3\n",
+    "keywords: [structured output, json, information extraction, with_structured_output]\n",
    "---"
   ]
  },
@@ -28,6 +33,8 @@
    "\n",
    "## The `.with_structured_output()` method\n",
    "\n",
+    "<span data-heading-keywords=\"with_structured_output\"></span>\n",
+    "\n",
    ":::info Supported models\n",
    "\n",
    "You can find a [list of models that support this method here](/docs/integrations/chat/).\n",
--- a/docs/docs/how_to/tool_calling.ipynb
+++ b/docs/docs/how_to/tool_calling.ipynb
@@ -167,13 +167,83 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 4,
+   "execution_count": 5,
   "metadata": {},
   "outputs": [],
   "source": [
    "llm_with_tools = llm.bind_tools(tools)"
   ]
  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "We can also use the `tool_choice` parameter to ensure certain behavior. For example, we can force our tool to call the multiply tool by using the following code:"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 9,
+   "metadata": {},
+   "outputs": [
+    {
+     "data": {
+      "text/plain": [
+       "AIMessage(content='', additional_kwargs={'tool_calls': [{'id': 'call_9cViskmLvPnHjXk9tbVla5HA', 'function': {'arguments': '{\"a\":2,\"b\":4}', 'name': 'Multiply'}, 'type': 'function'}]}, response_metadata={'token_usage': {'completion_tokens': 9, 'prompt_tokens': 103, 'total_tokens': 112}, 'model_name': 'gpt-3.5-turbo-0125', 'system_fingerprint': None, 'finish_reason': 'stop', 'logprobs': None}, id='run-095b827e-2bdd-43bb-8897-c843f4504883-0', tool_calls=[{'name': 'Multiply', 'args': {'a': 2, 'b': 4}, 'id': 'call_9cViskmLvPnHjXk9tbVla5HA'}], usage_metadata={'input_tokens': 103, 'output_tokens': 9, 'total_tokens': 112})"
+      ]
+     },
+     "execution_count": 9,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "llm_forced_to_multiply = llm.bind_tools(tools, tool_choice=\"Multiply\")\n",
+    "llm_forced_to_multiply.invoke(\"what is 2 + 4\")"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "Even if we pass it something that doesn't require multiplcation - it will still call the tool!"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "We can also just force our tool to select at least one of our tools by passing in the \"any\" (or \"required\" which is OpenAI specific) keyword to the `tool_choice` parameter."
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 10,
+   "metadata": {},
+   "outputs": [
+    {
+     "data": {
+      "text/plain": [
+       "AIMessage(content='', additional_kwargs={'tool_calls': [{'id': 'call_mCSiJntCwHJUBfaHZVUB2D8W', 'function': {'arguments': '{\"a\":1,\"b\":2}', 'name': 'Add'}, 'type': 'function'}]}, response_metadata={'token_usage': {'completion_tokens': 15, 'prompt_tokens': 94, 'total_tokens': 109}, 'model_name': 'gpt-3.5-turbo-0125', 'system_fingerprint': None, 'finish_reason': 'stop', 'logprobs': None}, id='run-28f75260-9900-4bed-8cd3-f1579abb65e5-0', tool_calls=[{'name': 'Add', 'args': {'a': 1, 'b': 2}, 'id': 'call_mCSiJntCwHJUBfaHZVUB2D8W'}], usage_metadata={'input_tokens': 94, 'output_tokens': 15, 'total_tokens': 109})"
+      ]
+     },
+     "execution_count": 10,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "llm_forced_to_use_tool = llm.bind_tools(tools, tool_choice=\"any\")\n",
+    "llm_forced_to_use_tool.invoke(\"What day is today?\")"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "As we can see, even though the prompt didn't really suggest a tool call, our LLM made one since it was forced to do so. You can look at the docs for [`bind_tool`](https://api.python.langchain.com/en/latest/chat_models/langchain_openai.chat_models.base.BaseChatOpenAI.html#langchain_openai.chat_models.base.BaseChatOpenAI.bind_tools) to learn about all the ways to customize how your LLM selects tools."
+   ]
+  },
  {
   "cell_type": "markdown",
   "metadata": {},
@@ -711,7 +781,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
-   "version": "3.9.1"
+   "version": "3.11.9"
  }
 },
 "nbformat": 4,
--- a/docs/docs/how_to/tool_runtime.ipynb
+++ b/docs/docs/how_to/tool_runtime.ipynb
@@ -0,0 +1,256 @@
+{
+ "cells": [
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "# How to pass run time values to a tool\n",
+    "\n",
+    ":::info Prerequisites\n",
+    "\n",
+    "This guide assumes familiarity with the following concepts:\n",
+    "- [Chat models](/docs/concepts/#chat-models)\n",
+    "- [LangChain Tools](/docs/concepts/#tools)\n",
+    "- [How to create tools](/docs/how_to/custom_tools)\n",
+    "- [How to use a model to call tools](https://python.langchain.com/v0.2/docs/how_to/tool_calling/)\n",
+    ":::\n",
+    "\n",
+    ":::{.callout-info} Supported models\n",
+    "\n",
+    "This how-to guide uses models with native tool calling capability.\n",
+    "You can find a [list of all models that support tool calling](/docs/integrations/chat/).\n",
+    "\n",
+    ":::\n",
+    "\n",
+    ":::{.callout-info} Using with LangGraph\n",
+    "\n",
+    "If you're using LangGraph, please refer to [this how-to guide](https://langchain-ai.github.io/langgraph/how-tos/pass-run-time-values-to-tools/)\n",
+    "which shows how to create an agent that keeps track of a given user's favorite pets.\n",
+    ":::\n",
+    "\n",
+    "You may need to bind values to a tool that are only known at runtime. For example, the tool logic may require using the ID of the user who made the request.\n",
+    "\n",
+    "Most of the time, such values should not be controlled by the LLM. In fact, allowing the LLM to control the user ID may lead to a security risk.\n",
+    "\n",
+    "Instead, the LLM should only control the parameters of the tool that are meant to be controlled by the LLM, while other parameters (such as user ID) should be fixed by the application logic.\n",
+    "\n",
+    "This how-to guide shows a simple design pattern that creates the tool dynamically at run time and binds to them appropriate values."
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "We can bind them to chat models as follows:\n",
+    "\n",
+    "```{=mdx}\n",
+    "import ChatModelTabs from \"@theme/ChatModelTabs\";\n",
+    "\n",
+    "<ChatModelTabs\n",
+    "  customVarName=\"llm\"\n",
+    "  fireworksParams={`model=\"accounts/fireworks/models/firefunction-v1\", temperature=0`}\n",
+    "/>\n",
+    "```"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 1,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "\n",
+      "\u001b[1m[\u001b[0m\u001b[34;49mnotice\u001b[0m\u001b[1;39;49m]\u001b[0m\u001b[39;49m A new release of pip is available: \u001b[0m\u001b[31;49m23.2.1\u001b[0m\u001b[39;49m -> \u001b[0m\u001b[32;49m24.0\u001b[0m\n",
+      "\u001b[1m[\u001b[0m\u001b[34;49mnotice\u001b[0m\u001b[1;39;49m]\u001b[0m\u001b[39;49m To update, run: \u001b[0m\u001b[32;49mpython -m pip install --upgrade pip\u001b[0m\n",
+      "Note: you may need to restart the kernel to use updated packages.\n"
+     ]
+    }
+   ],
+   "source": [
+    "# | output: false\n",
+    "# | echo: false\n",
+    "\n",
+    "%pip install -qU langchain langchain_openai\n",
+    "\n",
+    "import os\n",
+    "from getpass import getpass\n",
+    "\n",
+    "from langchain_openai import ChatOpenAI\n",
+    "\n",
+    "if \"OPENAI_API_KEY\" not in os.environ:\n",
+    "    os.environ[\"OPENAI_API_KEY\"] = getpass()\n",
+    "\n",
+    "llm = ChatOpenAI(model=\"gpt-3.5-turbo-0125\", temperature=0)"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "# Passing request time information\n",
+    "\n",
+    "The idea is to create the tool dynamically at request time, and bind to it the appropriate information. For example,\n",
+    "this information may be the user ID as resolved from the request itself."
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 2,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "from typing import List\n",
+    "\n",
+    "from langchain_core.output_parsers import JsonOutputParser\n",
+    "from langchain_core.tools import BaseTool, tool"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 3,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "user_to_pets = {}\n",
+    "\n",
+    "\n",
+    "def generate_tools_for_user(user_id: str) -> List[BaseTool]:\n",
+    "    \"\"\"Generate a set of tools that have a user id associated with them.\"\"\"\n",
+    "\n",
+    "    @tool\n",
+    "    def update_favorite_pets(pets: List[str]) -> None:\n",
+    "        \"\"\"Add the list of favorite pets.\"\"\"\n",
+    "        user_to_pets[user_id] = pets\n",
+    "\n",
+    "    @tool\n",
+    "    def delete_favorite_pets() -> None:\n",
+    "        \"\"\"Delete the list of favorite pets.\"\"\"\n",
+    "        if user_id in user_to_pets:\n",
+    "            del user_to_pets[user_id]\n",
+    "\n",
+    "    @tool\n",
+    "    def list_favorite_pets() -> None:\n",
+    "        \"\"\"List favorite pets if any.\"\"\"\n",
+    "        return user_to_pets.get(user_id, [])\n",
+    "\n",
+    "    return [update_favorite_pets, delete_favorite_pets, list_favorite_pets]"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "Verify that the tools work correctly"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 4,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "{'eugene': ['cat', 'dog']}\n",
+      "['cat', 'dog']\n"
+     ]
+    }
+   ],
+   "source": [
+    "update_pets, delete_pets, list_pets = generate_tools_for_user(\"eugene\")\n",
+    "update_pets.invoke({\"pets\": [\"cat\", \"dog\"]})\n",
+    "print(user_to_pets)\n",
+    "print(list_pets.invoke({}))"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 5,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "from langchain_core.prompts import ChatPromptTemplate\n",
+    "\n",
+    "\n",
+    "def handle_run_time_request(user_id: str, query: str):\n",
+    "    \"\"\"Handle run time request.\"\"\"\n",
+    "    tools = generate_tools_for_user(user_id)\n",
+    "    llm_with_tools = llm.bind_tools(tools)\n",
+    "    prompt = ChatPromptTemplate.from_messages(\n",
+    "        [(\"system\", \"You are a helpful assistant.\")],\n",
+    "    )\n",
+    "    chain = prompt | llm_with_tools\n",
+    "    return llm_with_tools.invoke(query)"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "This code will allow the LLM to invoke the tools, but the LLM is **unaware** of the fact that a **user ID** even exists!"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 6,
+   "metadata": {},
+   "outputs": [
+    {
+     "data": {
+      "text/plain": [
+       "[{'name': 'update_favorite_pets',\n",
+       "  'args': {'pets': ['cats', 'parrots']},\n",
+       "  'id': 'call_jJvjPXsNbFO5MMgW0q84iqCN'}]"
+      ]
+     },
+     "execution_count": 6,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "ai_message = handle_run_time_request(\n",
+    "    \"eugene\", \"my favorite animals are cats and parrots.\"\n",
+    ")\n",
+    "ai_message.tool_calls"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    ":::{.callout-important}\n",
+    "\n",
+    "Chat models only output requests to invoke tools, they don't actually invoke the underlying tools.\n",
+    "\n",
+    "To see how to invoke the tools, please refer to [how to use a model to call tools](https://python.langchain.com/v0.2/docs/how_to/tool_calling/).\n",
+    ":::"
+   ]
+  }
+ ],
+ "metadata": {
+  "kernelspec": {
+   "display_name": "Python 3 (ipykernel)",
+   "language": "python",
+   "name": "python3"
+  },
+  "language_info": {
+   "codemirror_mode": {
+    "name": "ipython",
+    "version": 3
+   },
+   "file_extension": ".py",
+   "mimetype": "text/x-python",
+   "name": "python",
+   "nbconvert_exporter": "python",
+   "pygments_lexer": "ipython3",
+   "version": "3.11.4"
+  }
+ },
+ "nbformat": 4,
+ "nbformat_minor": 4
+}
--- a/docs/docs/how_to/tools_builtin.ipynb
+++ b/docs/docs/how_to/tools_builtin.ipynb
@@ -36,7 +36,7 @@
    "\n",
    "When using 3rd party tools, make sure that you understand how the tool works, what permissions\n",
    "it has. Read over its documentation and check if anything is required from you\n",
-    "from a security point of view. Please see our [security](https://python.langchain.com/v0.1/docs/security/) \n",
+    "from a security point of view. Please see our [security](https://python.langchain.com/v0.2/docs/security/) \n",
    "guidelines for more information.\n",
    "\n",
    ":::\n",
--- a/docs/docs/integrations/callbacks/llmonitor.md
+++ b/docs/docs/integrations/callbacks/llmonitor.md
@@ -110,7 +110,7 @@ with identify("user-123"):
    llm.invoke("Tell me a joke")

 with identify("user-456", user_props={"email": "user456@test.com"}):
-    agen.run("Who is Leo DiCaprio's girlfriend?")
+    agent.run("Who is Leo DiCaprio's girlfriend?")
 ```
 ## Support

--- a/docs/docs/integrations/callbacks/upstash_ratelimit.ipynb
+++ b/docs/docs/integrations/callbacks/upstash_ratelimit.ipynb
@@ -0,0 +1,245 @@
+{
+ "cells": [
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "# Upstash Ratelimit Callback\n",
+    "\n",
+    "In this guide, we will go over how to add rate limiting based on number of requests or the number of tokens using `UpstashRatelimitHandler`. This handler uses [ratelimit library of Upstash](https://github.com/upstash/ratelimit-py/), which utilizes [Upstash Redis](https://upstash.com/docs/redis/overall/getstarted).\n",
+    "\n",
+    "Upstash Ratelimit works by sending an HTTP request to Upstash Redis everytime the `limit` method is called. Remaining tokens/requests of the user are checked and updated. Based on the remaining tokens, we can stop the execution of costly operations like invoking an LLM or querying a vector store:\n",
+    "\n",
+    "```py\n",
+    "response = ratelimit.limit()\n",
+    "if response.allowed:\n",
+    "    execute_costly_operation()\n",
+    "```\n",
+    "\n",
+    "`UpstashRatelimitHandler` allows you to incorporate the ratelimit logic into your chain in a few minutes.\n",
+    "\n",
+    "First, you will need to go to [the Upstash Console](https://console.upstash.com/login) and create a redis database ([see our docs](https://upstash.com/docs/redis/overall/getstarted)). After creating a database, you will need to set the environment variables:\n",
+    "\n",
+    "```\n",
+    "UPSTASH_REDIS_REST_URL=\"****\"\n",
+    "UPSTASH_REDIS_REST_TOKEN=\"****\"\n",
+    "```\n",
+    "\n",
+    "Next, you will need to install Upstash Ratelimit and Redis library with:\n",
+    "\n",
+    "```\n",
+    "pip install upstash-ratelimit upstash-redis\n",
+    "```\n",
+    "\n",
+    "You are now ready to add rate limiting to your chain!\n",
+    "\n",
+    "## Ratelimiting Per Request\n",
+    "\n",
+    "Let's imagine that we want to allow our users to invoke our chain 10 times per minute. Achieving this is as simple as:"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 21,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "Error in UpstashRatelimitHandler.on_chain_start callback: UpstashRatelimitError('Request limit reached!')\n"
+     ]
+    },
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "Handling ratelimit. <class 'langchain_community.callbacks.upstash_ratelimit_callback.UpstashRatelimitError'>\n"
+     ]
+    }
+   ],
+   "source": [
+    "# set env variables\n",
+    "import os\n",
+    "\n",
+    "os.environ[\"UPSTASH_REDIS_REST_URL\"] = \"****\"\n",
+    "os.environ[\"UPSTASH_REDIS_REST_TOKEN\"] = \"****\"\n",
+    "\n",
+    "from langchain_community.callbacks import UpstashRatelimitError, UpstashRatelimitHandler\n",
+    "from langchain_core.runnables import RunnableLambda\n",
+    "from upstash_ratelimit import FixedWindow, Ratelimit\n",
+    "from upstash_redis import Redis\n",
+    "\n",
+    "# create ratelimit\n",
+    "ratelimit = Ratelimit(\n",
+    "    redis=Redis.from_env(),\n",
+    "    # 10 requests per window, where window size is 60 seconds:\n",
+    "    limiter=FixedWindow(max_requests=10, window=60),\n",
+    ")\n",
+    "\n",
+    "# create handler\n",
+    "user_id = \"user_id\"  # should be a method which gets the user id\n",
+    "handler = UpstashRatelimitHandler(identifier=user_id, request_ratelimit=ratelimit)\n",
+    "\n",
+    "# create mock chain\n",
+    "chain = RunnableLambda(str)\n",
+    "\n",
+    "# invoke chain with handler:\n",
+    "try:\n",
+    "    result = chain.invoke(\"Hello world!\", config={\"callbacks\": [handler]})\n",
+    "except UpstashRatelimitError:\n",
+    "    print(\"Handling ratelimit.\", UpstashRatelimitError)"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "Note that we pass the handler to the `invoke` method instead of passing the handler when defining the chain.\n",
+    "\n",
+    "For rate limiting algorithms other than `FixedWindow`, see [upstash-ratelimit docs](https://github.com/upstash/ratelimit-py?tab=readme-ov-file#ratelimiting-algorithms).\n",
+    "\n",
+    "Before executing any steps in our pipeline, ratelimit will check whether the user has passed the request limit. If so, `UpstashRatelimitError` is raised.\n",
+    "\n",
+    "## Ratelimiting Per Token\n",
+    "\n",
+    "Another option is to rate limit chain invokations based on:\n",
+    "1. number of tokens in prompt\n",
+    "2. number of tokens in prompt and LLM completion\n",
+    "\n",
+    "This only works if you have an LLM in your chain. Another requirement is that the LLM you are using should return the token usage in it's `LLMOutput`.\n",
+    "\n",
+    "### How it works\n",
+    "\n",
+    "The handler will get the remaining tokens before calling the LLM. If the remaining tokens is more than 0, LLM will be called. Otherwise `UpstashRatelimitError` will be raised.\n",
+    "\n",
+    "After LLM is called, token usage information will be used to subtracted from the remaining tokens of the user. No error is raised at this stage of the chain.\n",
+    "\n",
+    "### Configuration\n",
+    "\n",
+    "For the first configuration, simply initialize the handler like this:"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "ratelimit = Ratelimit(\n",
+    "    redis=Redis.from_env(),\n",
+    "    # 1000 tokens per window, where window size is 60 seconds:\n",
+    "    limiter=FixedWindow(max_requests=1000, window=60),\n",
+    ")\n",
+    "\n",
+    "handler = UpstashRatelimitHandler(identifier=user_id, token_ratelimit=ratelimit)"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "For the second configuration, here is how to initialize the handler:"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "ratelimit = Ratelimit(\n",
+    "    redis=Redis.from_env(),\n",
+    "    # 1000 tokens per window, where window size is 60 seconds:\n",
+    "    limiter=FixedWindow(max_requests=1000, window=60),\n",
+    ")\n",
+    "\n",
+    "handler = UpstashRatelimitHandler(\n",
+    "    identifier=user_id,\n",
+    "    token_ratelimit=ratelimit,\n",
+    "    include_output_tokens=True,  # set to True\n",
+    ")"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "You can also employ ratelimiting based on requests and tokens at the same time, simply by passing both `request_ratelimit` and `token_ratelimit` parameters.\n",
+    "\n",
+    "Here is an example with a chain utilizing an LLM:"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 22,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "Error in UpstashRatelimitHandler.on_llm_start callback: UpstashRatelimitError('Token limit reached!')\n"
+     ]
+    },
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "Handling ratelimit. <class 'langchain_community.callbacks.upstash_ratelimit_callback.UpstashRatelimitError'>\n"
+     ]
+    }
+   ],
+   "source": [
+    "# set env variables\n",
+    "import os\n",
+    "\n",
+    "os.environ[\"UPSTASH_REDIS_REST_URL\"] = \"****\"\n",
+    "os.environ[\"UPSTASH_REDIS_REST_TOKEN\"] = \"****\"\n",
+    "os.environ[\"OPENAI_API_KEY\"] = \"****\"\n",
+    "\n",
+    "from langchain_community.callbacks import UpstashRatelimitError, UpstashRatelimitHandler\n",
+    "from langchain_core.runnables import RunnableLambda\n",
+    "from langchain_openai import ChatOpenAI\n",
+    "from upstash_ratelimit import FixedWindow, Ratelimit\n",
+    "from upstash_redis import Redis\n",
+    "\n",
+    "# create ratelimit\n",
+    "ratelimit = Ratelimit(\n",
+    "    redis=Redis.from_env(),\n",
+    "    # 500 tokens per window, where window size is 60 seconds:\n",
+    "    limiter=FixedWindow(max_requests=500, window=60),\n",
+    ")\n",
+    "\n",
+    "# create handler\n",
+    "user_id = \"user_id\"  # should be a method which gets the user id\n",
+    "handler = UpstashRatelimitHandler(identifier=user_id, token_ratelimit=ratelimit)\n",
+    "\n",
+    "# create mock chain\n",
+    "as_str = RunnableLambda(str)\n",
+    "model = ChatOpenAI()\n",
+    "\n",
+    "chain = as_str | model\n",
+    "\n",
+    "# invoke chain with handler:\n",
+    "try:\n",
+    "    result = chain.invoke(\"Hello world!\", config={\"callbacks\": [handler]})\n",
+    "except UpstashRatelimitError:\n",
+    "    print(\"Handling ratelimit.\", UpstashRatelimitError)"
+   ]
+  }
+ ],
+ "metadata": {
+  "kernelspec": {
+   "display_name": "lc39",
+   "language": "python",
+   "name": "python3"
+  },
+  "language_info": {
+   "name": "python",
+   "version": "3.9.19"
+  }
+ },
+ "nbformat": 4,
+ "nbformat_minor": 2
+}
--- a/docs/docs/integrations/chat/anthropic.ipynb
+++ b/docs/docs/integrations/chat/anthropic.ipynb
--- a/docs/docs/integrations/chat/azure_chat_openai.ipynb
+++ b/docs/docs/integrations/chat/azure_chat_openai.ipynb
@@ -23,13 +23,11 @@
   ]
  },
  {
-   "cell_type": "raw",
+   "cell_type": "code",
+   "execution_count": null,
   "id": "d83ba7de",
-   "metadata": {
-    "vscode": {
-     "languageId": "raw"
-    }
-   },
+   "metadata": {},
+   "outputs": [],
   "source": [
    "%pip install -qU langchain-openai"
   ]
--- a/docs/docs/integrations/chat/cohere.ipynb
+++ b/docs/docs/integrations/chat/cohere.ipynb
@@ -201,7 +201,7 @@
   "source": [
    "## Chaining\n",
    "\n",
-    "You can also easily combine with a prompt template for easy structuring of user input. We can do this using [LCEL](/docs/concepts#langchain-expression-language)"
+    "You can also easily combine with a prompt template for easy structuring of user input. We can do this using [LCEL](/docs/concepts#langchain-expression-language-lcel)"
   ]
  },
  {
--- a/docs/docs/integrations/chat/google_vertex_ai_palm.ipynb
+++ b/docs/docs/integrations/chat/google_vertex_ai_palm.ipynb
@@ -2,33 +2,50 @@
 "cells": [
  {
   "cell_type": "raw",
+   "id": "afaf8039",
   "metadata": {},
   "source": [
    "---\n",
    "sidebar_label: Google Cloud Vertex AI\n",
-    "keywords: [gemini, vertex, ChatVertexAI, gemini-pro]\n",
    "---"
   ]
  },
  {
   "cell_type": "markdown",
+   "id": "e49f1e0d",
   "metadata": {},
   "source": [
    "# ChatVertexAI\n",
    "\n",
-    "Note: This is separate from the Google PaLM integration. Google has chosen to offer an enterprise version of PaLM through GCP, and this supports the models made available through there. \n",
+    "This page provides a quick overview for getting started with VertexAI [chat models](/docs/concepts/#chat-models). For detailed documentation of all ChatVertexAI features and configurations head to the [API reference](https://api.python.langchain.com/en/latest/chat_models/langchain_google_vertexai.chat_models.ChatVertexAI.html).\n",
    "\n",
-    "ChatVertexAI exposes all foundational models available in Google Cloud:\n",
+    "ChatVertexAI exposes all foundational models available in Google Cloud, like `gemini-1.5-pro`, `gemini-1.5-flash`, etc. For a full and updated list of available models visit [VertexAI documentation](https://cloud.google.com/vertex-ai/docs/generative-ai/model-reference/overview).\n",
    "\n",
-    "- Gemini (`gemini-pro` and `gemini-pro-vision`)\n",
-    "- PaLM 2 for Text (`text-bison`)\n",
-    "- Codey for Code Generation (`codechat-bison`)\n",
+    ":::info Google Cloud VertexAI vs Google PaLM\n",
    "\n",
-    "For a full and updated list of available models visit [VertexAI documentation](https://cloud.google.com/vertex-ai/docs/generative-ai/model-reference/overview).\n",
+    "The Google Cloud VertexAI integration is separate from the [Google PaLM integration](/docs/integrations/chat/google_generative_ai/). Google has chosen to offer an enterprise version of PaLM through GCP, and this supports the models made available through there. \n",
    "\n",
-    "By default, Google Cloud [does not use](https://cloud.google.com/vertex-ai/docs/generative-ai/data-governance#foundation_model_development) customer data to train its foundation models as part of Google Cloud`s AI/ML Privacy Commitment. More details about how Google processes data can also be found in [Google's Customer Data Processing Addendum (CDPA)](https://cloud.google.com/terms/data-processing-addendum).\n",
+    ":::\n",
    "\n",
-    "To use `Google Cloud Vertex AI` PaLM you must have the `langchain-google-vertexai` Python package installed and either:\n",
+    "## Overview\n",
+    "### Integration details\n",
+    "\n",
+    "| Class | Package | Local | Serializable | [JS support](https://js.langchain.com/v0.2/docs/integrations/chat/google_vertex_ai) | Package downloads | Package latest |\n",
+    "| :--- | :--- | :---: | :---: |  :---: | :---: | :---: |\n",
+    "| [ChatVertexAI](https://api.python.langchain.com/en/latest/chat_models/langchain_google_vertexai.chat_models.ChatVertexAI.html) | [langchain-google-vertexai](https://api.python.langchain.com/en/latest/google_vertexai_api_reference.html) | ❌ | beta | ✅ | ![PyPI - Downloads](https://img.shields.io/pypi/dm/langchain-google-vertexai?style=flat-square&label=%20) | ![PyPI - Version](https://img.shields.io/pypi/v/langchain-google-vertexai?style=flat-square&label=%20) |\n",
+    "\n",
+    "### Model features\n",
+    "| [Tool calling](/docs/how_to/tool_calling/) | [Structured output](/docs/how_to/structured_output/) | JSON mode | [Image input](/docs/how_to/multimodal_inputs/) | Audio input | Video input | [Token-level streaming](/docs/how_to/chat_streaming/) | Native async | [Token usage](/docs/how_to/chat_token_usage_tracking/) | [Logprobs](/docs/how_to/logprobs/) |\n",
+    "| :---: | :---: | :---: | :---: |  :---: | :---: | :---: | :---: | :---: | :---: |\n",
+    "| ✅ | ✅ | ❌ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | \n",
+    "\n",
+    "## Setup\n",
+    "\n",
+    "To access VertexAI models you'll need to create a Google Cloud Platform account, set up credentials, and install the `langchain-google-vertexai` integration package.\n",
+    "\n",
+    "### Credentials\n",
+    "\n",
+    "To use the integration you must:\n",
    "- Have credentials configured for your environment (gcloud, workload identity, etc...)\n",
    "- Store the path to a service account JSON file as the GOOGLE_APPLICATION_CREDENTIALS environment variable\n",
    "\n",
@@ -37,432 +54,156 @@
    "For more information, see: \n",
    "- https://cloud.google.com/docs/authentication/application-default-credentials#GAC\n",
    "- https://googleapis.dev/python/google-auth/latest/reference/google.auth.html#module-google.auth\n",
-    "\n"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "%pip install --upgrade --quiet  langchain-google-vertexai"
+    "\n",
+    "If you want to get automated tracing of your model calls you can also set your [LangSmith](https://docs.smith.langchain.com/) API key by uncommenting below:"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 1,
+   "id": "a15d341e-3e26-4ca3-830b-5aab30ed66de",
   "metadata": {},
   "outputs": [],
   "source": [
-    "from langchain_core.prompts import ChatPromptTemplate\n",
-    "from langchain_google_vertexai import ChatVertexAI"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "AIMessage(content=\" J'aime la programmation.\")"
-      ]
-     },
-     "execution_count": null,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "system = \"You are a helpful assistant who translate English to French\"\n",
-    "human = \"Translate this sentence from English to French. I love programming.\"\n",
-    "prompt = ChatPromptTemplate.from_messages([(\"system\", system), (\"human\", human)])\n",
-    "\n",
-    "chat = ChatVertexAI()\n",
-    "\n",
-    "chain = prompt | chat\n",
-    "chain.invoke({})"
+    "# os.environ[\"LANGSMITH_API_KEY\"] = getpass.getpass(\"Enter your LangSmith API key: \")\n",
+    "# os.environ[\"LANGSMITH_TRACING\"] = \"true\""
   ]
  },
  {
   "cell_type": "markdown",
+   "id": "0730d6a1-c893-4840-9817-5e5251676d5d",
   "metadata": {},
   "source": [
-    "Gemini doesn't support SystemMessage at the moment, but it can be added to the first human message in the row. If you want such behavior, just set the `convert_system_message_to_human` to `True`:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "AIMessage(content=\"J'aime la programmation.\")"
-      ]
-     },
-     "execution_count": null,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "system = \"You are a helpful assistant who translate English to French\"\n",
-    "human = \"Translate this sentence from English to French. I love programming.\"\n",
-    "prompt = ChatPromptTemplate.from_messages([(\"system\", system), (\"human\", human)])\n",
+    "### Installation\n",
    "\n",
-    "chat = ChatVertexAI(model=\"gemini-pro\", convert_system_message_to_human=True)\n",
-    "\n",
-    "chain = prompt | chat\n",
-    "chain.invoke({})"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "If we want to construct a simple chain that takes user specified parameters:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "AIMessage(content=' プログラミングが大好きです')"
-      ]
-     },
-     "execution_count": null,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "system = (\n",
-    "    \"You are a helpful assistant that translates {input_language} to {output_language}.\"\n",
-    ")\n",
-    "human = \"{text}\"\n",
-    "prompt = ChatPromptTemplate.from_messages([(\"system\", system), (\"human\", human)])\n",
-    "\n",
-    "chat = ChatVertexAI()\n",
-    "\n",
-    "chain = prompt | chat\n",
-    "\n",
-    "chain.invoke(\n",
-    "    {\n",
-    "        \"input_language\": \"English\",\n",
-    "        \"output_language\": \"Japanese\",\n",
-    "        \"text\": \"I love programming\",\n",
-    "    }\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "## Code generation chat models\n",
-    "You can now leverage the Codey API for code chat within Vertex AI. The model available is:\n",
-    "- `codechat-bison`: for code assistance"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      " ```python\n",
-      "def is_prime(n):\n",
-      "  \"\"\"\n",
-      "  Check if a number is prime.\n",
-      "\n",
-      "  Args:\n",
-      "    n: The number to check.\n",
-      "\n",
-      "  Returns:\n",
-      "    True if n is prime, False otherwise.\n",
-      "  \"\"\"\n",
-      "\n",
-      "  # If n is 1, it is not prime.\n",
-      "  if n == 1:\n",
-      "    return False\n",
-      "\n",
-      "  # Iterate over all numbers from 2 to the square root of n.\n",
-      "  for i in range(2, int(n ** 0.5) + 1):\n",
-      "    # If n is divisible by any number from 2 to its square root, it is not prime.\n",
-      "    if n % i == 0:\n",
-      "      return False\n",
-      "\n",
-      "  # If n is divisible by no number from 2 to its square root, it is prime.\n",
-      "  return True\n",
-      "\n",
-      "\n",
-      "def find_prime_numbers(n):\n",
-      "  \"\"\"\n",
-      "  Find all prime numbers up to a given number.\n",
-      "\n",
-      "  Args:\n",
-      "    n: The upper bound for the prime numbers to find.\n",
-      "\n",
-      "  Returns:\n",
-      "    A list of all prime numbers up to n.\n",
-      "  \"\"\"\n",
-      "\n",
-      "  # Create a list of all numbers from 2 to n.\n",
-      "  numbers = list(range(2, n + 1))\n",
-      "\n",
-      "  # Iterate over the list of numbers and remove any that are not prime.\n",
-      "  for number in numbers:\n",
-      "    if not is_prime(number):\n",
-      "      numbers.remove(number)\n",
-      "\n",
-      "  # Return the list of prime numbers.\n",
-      "  return numbers\n",
-      "```\n"
-     ]
-    }
-   ],
-   "source": [
-    "chat = ChatVertexAI(model=\"codechat-bison\", max_tokens=1000, temperature=0.5)\n",
-    "\n",
-    "message = chat.invoke(\"Write a Python function generating all prime numbers\")\n",
-    "print(message.content)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "## Full generation info\n",
-    "\n",
-    "We can use the `generate` method to get back extra metadata like [safety attributes](https://cloud.google.com/vertex-ai/docs/generative-ai/learn/responsible-ai#safety_attribute_confidence_scoring) and not just chat completions\n",
-    "\n",
-    "Note that the `generation_info` will be different depending if you're using a gemini model or not.\n",
-    "\n",
-    "### Gemini model\n",
-    "\n",
-    "`generation_info` will include:\n",
-    "\n",
-    "- `is_blocked`: whether generation was blocked or not\n",
-    "- `safety_ratings`: safety ratings' categories and probability labels"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 4,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "from pprint import pprint\n",
-    "\n",
-    "from langchain_core.messages import HumanMessage\n",
-    "from langchain_google_vertexai import HarmBlockThreshold, HarmCategory"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 5,
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "{'citation_metadata': None,\n",
-      " 'is_blocked': False,\n",
-      " 'safety_ratings': [{'blocked': False,\n",
-      "                     'category': 'HARM_CATEGORY_HATE_SPEECH',\n",
-      "                     'probability_label': 'NEGLIGIBLE'},\n",
-      "                    {'blocked': False,\n",
-      "                     'category': 'HARM_CATEGORY_DANGEROUS_CONTENT',\n",
-      "                     'probability_label': 'NEGLIGIBLE'},\n",
-      "                    {'blocked': False,\n",
-      "                     'category': 'HARM_CATEGORY_HARASSMENT',\n",
-      "                     'probability_label': 'NEGLIGIBLE'},\n",
-      "                    {'blocked': False,\n",
-      "                     'category': 'HARM_CATEGORY_SEXUALLY_EXPLICIT',\n",
-      "                     'probability_label': 'NEGLIGIBLE'}],\n",
-      " 'usage_metadata': {'candidates_token_count': 6,\n",
-      "                    'prompt_token_count': 12,\n",
-      "                    'total_token_count': 18}}\n"
-     ]
-    }
-   ],
-   "source": [
-    "human = \"Translate this sentence from English to French. I love programming.\"\n",
-    "messages = [HumanMessage(content=human)]\n",
-    "\n",
-    "\n",
-    "chat = ChatVertexAI(\n",
-    "    model_name=\"gemini-pro\",\n",
-    "    safety_settings={\n",
-    "        HarmCategory.HARM_CATEGORY_HATE_SPEECH: HarmBlockThreshold.BLOCK_LOW_AND_ABOVE\n",
-    "    },\n",
-    ")\n",
-    "\n",
-    "result = chat.generate([messages])\n",
-    "pprint(result.generations[0][0].generation_info)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "### Non-gemini model\n",
-    "\n",
-    "`generation_info` will include:\n",
-    "\n",
-    "- `is_blocked`: whether generation was blocked or not\n",
-    "- `safety_attributes`: a dictionary mapping safety attributes to their scores"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 6,
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "{'errors': (),\n",
-      " 'grounding_metadata': {'citations': [], 'search_queries': []},\n",
-      " 'is_blocked': False,\n",
-      " 'safety_attributes': [{'Derogatory': 0.1, 'Insult': 0.1, 'Sexual': 0.2}],\n",
-      " 'usage_metadata': {'candidates_billable_characters': 88.0,\n",
-      "                    'candidates_token_count': 24.0,\n",
-      "                    'prompt_billable_characters': 58.0,\n",
-      "                    'prompt_token_count': 12.0}}\n"
-     ]
-    }
-   ],
-   "source": [
-    "chat = ChatVertexAI()  # default is `chat-bison`\n",
-    "\n",
-    "result = chat.generate([messages])\n",
-    "pprint(result.generations[0][0].generation_info)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "## Tool calling (a.k.a. function calling) with Gemini\n",
-    "\n",
-    "We can pass tool definitions to Gemini models to get the model to invoke those tools when appropriate. This is useful not only for LLM-powered tool use but also for getting structured outputs out of models more generally.\n",
-    "\n",
-    "With `ChatVertexAI.bind_tools()`, we can easily pass in Pydantic classes, dict schemas, LangChain tools, or even functions as tools to the model. Under the hood these are converted to a Gemini tool schema, which looks like:\n",
-    "```python\n",
-    "{\n",
-    "    \"name\": \"...\",  # tool name\n",
-    "    \"description\": \"...\",  # tool description\n",
-    "    \"parameters\": {...}  # tool input schema as JSONSchema\n",
-    "}\n",
-    "```"
+    "The LangChain VertexAI integration lives in the `langchain-google-vertexai` package:"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 2,
+   "id": "652d6238-1f87-422a-b135-f5abbb8652fc",
   "metadata": {},
   "outputs": [
    {
-     "data": {
-      "text/plain": [
-       "AIMessage(content='', additional_kwargs={'function_call': {'name': 'GetWeather', 'arguments': '{\"location\": \"San Francisco, CA\"}'}}, response_metadata={'is_blocked': False, 'safety_ratings': [{'category': 'HARM_CATEGORY_HATE_SPEECH', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_DANGEROUS_CONTENT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_HARASSMENT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_SEXUALLY_EXPLICIT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}], 'citation_metadata': None, 'usage_metadata': {'prompt_token_count': 41, 'candidates_token_count': 7, 'total_token_count': 48}}, id='run-05e760dc-0682-4286-88e1-5b23df69b083-0', tool_calls=[{'name': 'GetWeather', 'args': {'location': 'San Francisco, CA'}, 'id': 'cd2499c4-4513-4059-bfff-5321b6e922d0'}])"
-      ]
-     },
-     "execution_count": 2,
-     "metadata": {},
-     "output_type": "execute_result"
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "Note: you may need to restart the kernel to use updated packages.\n"
+     ]
    }
   ],
   "source": [
-    "from langchain.pydantic_v1 import BaseModel, Field\n",
-    "\n",
-    "\n",
-    "class GetWeather(BaseModel):\n",
-    "    \"\"\"Get the current weather in a given location\"\"\"\n",
-    "\n",
-    "    location: str = Field(..., description=\"The city and state, e.g. San Francisco, CA\")\n",
-    "\n",
-    "\n",
-    "llm = ChatVertexAI(model=\"gemini-pro\", temperature=0)\n",
-    "llm_with_tools = llm.bind_tools([GetWeather])\n",
-    "ai_msg = llm_with_tools.invoke(\n",
-    "    \"what is the weather like in San Francisco\",\n",
-    ")\n",
-    "ai_msg"
+    "%pip install -qU langchain-google-vertexai"
   ]
  },
  {
   "cell_type": "markdown",
+   "id": "a38cde65-254d-4219-a441-068766c0d4b5",
   "metadata": {},
   "source": [
-    "The tool calls can be access via the `AIMessage.tool_calls` attribute, where they are extracted in a model-agnostic format:"
+    "## Instantiation\n",
+    "\n",
+    "Now we can instantiate our model object and generate chat completions:"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 3,
+   "id": "cb09c344-1836-4e0c-acf8-11d13ac1dbae",
   "metadata": {},
+   "outputs": [],
+   "source": [
+    "from langchain_google_vertexai import ChatVertexAI\n",
+    "\n",
+    "llm = ChatVertexAI(\n",
+    "    model=\"gemini-1.5-flash-001\",\n",
+    "    temperature=0,\n",
+    "    max_tokens=None,\n",
+    "    max_retries=6,\n",
+    "    stop=None,\n",
+    "    # other params...\n",
+    ")"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "2b4f3e15",
+   "metadata": {},
+   "source": [
+    "## Invocation"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 4,
+   "id": "62e0dbc3",
+   "metadata": {
+    "tags": []
+   },
   "outputs": [
    {
     "data": {
      "text/plain": [
-       "[{'name': 'GetWeather',\n",
-       "  'args': {'location': 'San Francisco, CA'},\n",
-       "  'id': 'cd2499c4-4513-4059-bfff-5321b6e922d0'}]"
+       "AIMessage(content=\"J'adore programmer. \\n\", response_metadata={'is_blocked': False, 'safety_ratings': [{'category': 'HARM_CATEGORY_HATE_SPEECH', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_DANGEROUS_CONTENT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_HARASSMENT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_SEXUALLY_EXPLICIT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}], 'usage_metadata': {'prompt_token_count': 20, 'candidates_token_count': 7, 'total_token_count': 27}}, id='run-7032733c-d05c-4f0c-a17a-6c575fdd1ae0-0', usage_metadata={'input_tokens': 20, 'output_tokens': 7, 'total_tokens': 27})"
      ]
     },
-     "execution_count": 3,
+     "execution_count": 4,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
-    "ai_msg.tool_calls"
+    "messages = [\n",
+    "    (\n",
+    "        \"system\",\n",
+    "        \"You are a helpful assistant that translates English to French. Translate the user sentence.\",\n",
+    "    ),\n",
+    "    (\"human\", \"I love programming.\"),\n",
+    "]\n",
+    "ai_msg = llm.invoke(messages)\n",
+    "ai_msg"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 5,
+   "id": "d86145b3-bfef-46e8-b227-4dda5c9c2705",
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "J'adore programmer. \n",
+      "\n"
+     ]
+    }
+   ],
+   "source": [
+    "print(ai_msg.content)"
   ]
  },
  {
   "cell_type": "markdown",
+   "id": "18e2bfc0-7e78-4528-a73f-499ac150dca8",
   "metadata": {},
   "source": [
-    "For a complete guide on tool calling [head here](/docs/how_to/function_calling)."
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "## Structured outputs\n",
+    "## Chaining\n",
    "\n",
-    "Many applications require structured model outputs. Tool calling makes it much easier to do this reliably. The [with_structured_outputs](https://api.python.langchain.com/en/latest/chat_models/langchain_google_vertexai.chat_models.ChatVertexAI.html) constructor provides a simple interface built on top of tool calling for getting structured outputs out of a model. For a complete guide on structured outputs [head here](/docs/how_to/structured_output).\n",
-    "\n",
-    "###  ChatVertexAI.with_structured_outputs()\n",
-    "\n",
-    "To get structured outputs from our Gemini model all we need to do is to specify a desired schema, either as a Pydantic class or as a JSON schema, "
+    "We can [chain](/docs/how_to/sequence/) our model with a prompt template like so:"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 6,
+   "id": "e197d1d7-a070-4c96-9f8a-a0e86d046e0b",
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
-       "Person(name='Stefan', age=13)"
+       "AIMessage(content='Ich liebe Programmieren. \\n', response_metadata={'is_blocked': False, 'safety_ratings': [{'category': 'HARM_CATEGORY_HATE_SPEECH', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_DANGEROUS_CONTENT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_HARASSMENT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_SEXUALLY_EXPLICIT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}], 'usage_metadata': {'prompt_token_count': 15, 'candidates_token_count': 8, 'total_token_count': 23}}, id='run-c71955fd-8dc1-422b-88a7-853accf4811b-0', usage_metadata={'input_tokens': 15, 'output_tokens': 8, 'total_tokens': 23})"
      ]
     },
     "execution_count": 6,
@@ -471,139 +212,36 @@
    }
   ],
   "source": [
-    "class Person(BaseModel):\n",
-    "    \"\"\"Save information about a person.\"\"\"\n",
+    "from langchain_core.prompts import ChatPromptTemplate\n",
    "\n",
-    "    name: str = Field(..., description=\"The person's name.\")\n",
-    "    age: int = Field(..., description=\"The person's age.\")\n",
-    "\n",
-    "\n",
-    "structured_llm = llm.with_structured_output(Person)\n",
-    "structured_llm.invoke(\"Stefan is already 13 years old\")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "### [Legacy] Using `create_structured_runnable()`\n",
-    "\n",
-    "The legacy wasy to get structured outputs is using the `create_structured_runnable` constructor:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "from langchain_google_vertexai import create_structured_runnable\n",
-    "\n",
-    "chain = create_structured_runnable(Person, llm)\n",
-    "chain.invoke(\"My name is Erick and I'm 27 years old\")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "## Asynchronous calls\n",
-    "\n",
-    "We can make asynchronous calls via the Runnables [Async Interface](/docs/concepts#interface)."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "# for running these examples in the notebook:\n",
-    "import asyncio\n",
-    "\n",
-    "import nest_asyncio\n",
-    "\n",
-    "nest_asyncio.apply()"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "AIMessage(content=' अहं प्रोग्रामनं प्रेमामि')"
-      ]
-     },
-     "execution_count": null,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "system = (\n",
-    "    \"You are a helpful assistant that translates {input_language} to {output_language}.\"\n",
+    "prompt = ChatPromptTemplate.from_messages(\n",
+    "    [\n",
+    "        (\n",
+    "            \"system\",\n",
+    "            \"You are a helpful assistant that translates {input_language} to {output_language}.\",\n",
+    "        ),\n",
+    "        (\"human\", \"{input}\"),\n",
+    "    ]\n",
    ")\n",
-    "human = \"{text}\"\n",
-    "prompt = ChatPromptTemplate.from_messages([(\"system\", system), (\"human\", human)])\n",
    "\n",
-    "chat = ChatVertexAI(model=\"chat-bison\", max_tokens=1000, temperature=0.5)\n",
-    "chain = prompt | chat\n",
-    "\n",
-    "asyncio.run(\n",
-    "    chain.ainvoke(\n",
-    "        {\n",
-    "            \"input_language\": \"English\",\n",
-    "            \"output_language\": \"Sanskrit\",\n",
-    "            \"text\": \"I love programming\",\n",
-    "        }\n",
-    "    )\n",
+    "chain = prompt | llm\n",
+    "chain.invoke(\n",
+    "    {\n",
+    "        \"input_language\": \"English\",\n",
+    "        \"output_language\": \"German\",\n",
+    "        \"input\": \"I love programming.\",\n",
+    "    }\n",
    ")"
   ]
  },
  {
   "cell_type": "markdown",
+   "id": "3a5bb5ca-c3ae-4a58-be67-2cd18574b9a3",
   "metadata": {},
   "source": [
-    "## Streaming calls\n",
+    "## API reference\n",
    "\n",
-    "We can also stream outputs via the `stream` method:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      " The five most populous countries in the world are:\n",
-      "1. China (1.4 billion)\n",
-      "2. India (1.3 billion)\n",
-      "3. United States (331 million)\n",
-      "4. Indonesia (273 million)\n",
-      "5. Pakistan (220 million)"
-     ]
-    }
-   ],
-   "source": [
-    "import sys\n",
-    "\n",
-    "prompt = ChatPromptTemplate.from_messages(\n",
-    "    [(\"human\", \"List out the 5 most populous countries in the world\")]\n",
-    ")\n",
-    "\n",
-    "chat = ChatVertexAI()\n",
-    "\n",
-    "chain = prompt | chat\n",
-    "\n",
-    "for chunk in chain.stream({}):\n",
-    "    sys.stdout.write(chunk.content)\n",
-    "    sys.stdout.flush()"
+    "For detailed documentation of all ChatVertexAI features and configurations, like how to send multimodal inputs and configure safety settings, head to the API reference: https://api.python.langchain.com/en/latest/chat_models/langchain_google_vertexai.chat_models.ChatVertexAI.html"
   ]
  }
 ],
@@ -627,5 +265,5 @@
  }
 },
 "nbformat": 4,
- "nbformat_minor": 4
+ "nbformat_minor": 5
 }
--- a/docs/docs/integrations/chat/groq.ipynb
+++ b/docs/docs/integrations/chat/groq.ipynb
@@ -2,10 +2,15 @@
 "cells": [
  {
   "cell_type": "raw",
-   "metadata": {},
+   "metadata": {
+    "vscode": {
+     "languageId": "raw"
+    }
+   },
   "source": [
    "---\n",
    "sidebar_label: Groq\n",
+    "keywords: [chatgroq]\n",
    "---"
   ]
  },
@@ -15,45 +20,67 @@
   "source": [
    "# Groq\n",
    "\n",
-    "Install the langchain-groq package if not already installed:\n",
+    "LangChain supports integration with [Groq](https://groq.com/) chat models. Groq specializes in fast AI inference.\n",
    "\n",
-    "```bash\n",
-    "pip install langchain-groq\n",
-    "```\n",
-    "\n",
-    "Request an [API key](https://wow.groq.com) and set it as an environment variable:\n",
-    "\n",
-    "```bash\n",
-    "export GROQ_API_KEY=<YOUR API KEY>\n",
-    "```\n",
-    "\n",
-    "Alternatively, you may configure the API key when you initialize ChatGroq."
+    "To get started, you'll first need to install the langchain-groq package:"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "%pip install -qU langchain-groq"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
-    "Import the ChatGroq class and initialize it with a model:"
+    "Request an [API key](https://wow.groq.com) and set it as an environment variable:\n",
+    "\n",
+    "```bash\n",
+    "export GROQ_API_KEY=<YOUR API KEY>\n",
+    "```\n",
+    "\n",
+    "Alternatively, you may configure the API key when you initialize ChatGroq.\n",
+    "\n",
+    "Here's an example of it in action:"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 27,
+   "execution_count": 8,
   "metadata": {},
-   "outputs": [],
+   "outputs": [
+    {
+     "data": {
+      "text/plain": [
+       "AIMessage(content=\"Low latency is crucial for Large Language Models (LLMs) because it directly impacts the user experience, model performance, and overall efficiency. Here are some reasons why low latency is essential for LLMs:\\n\\n1. **Real-time Interaction**: LLMs are often used in applications that require real-time interaction, such as chatbots, virtual assistants, and language translation. Low latency ensures that the model responds quickly to user input, providing a seamless and engaging experience.\\n2. **Conversational Flow**: In conversational AI, latency can disrupt the natural flow of conversation. Low latency helps maintain a smooth conversation, allowing users to respond quickly and naturally, without feeling like they're waiting for the model to catch up.\\n3. **Model Performance**: High latency can lead to increased error rates, as the model may struggle to keep up with the input pace. Low latency enables the model to process information more efficiently, resulting in better accuracy and performance.\\n4. **Scalability**: As the number of users and requests increases, low latency becomes even more critical. It allows the model to handle a higher volume of requests without sacrificing performance, making it more scalable and efficient.\\n5. **Resource Utilization**: Low latency can reduce the computational resources required to process requests. By minimizing latency, you can optimize resource allocation, reduce costs, and improve overall system efficiency.\\n6. **User Experience**: High latency can lead to frustration, abandonment, and a poor user experience. Low latency ensures that users receive timely responses, which is essential for building trust and satisfaction.\\n7. **Competitive Advantage**: In applications like customer service or language translation, low latency can be a key differentiator. It can provide a competitive advantage by offering a faster and more responsive experience, setting your application apart from others.\\n8. **Edge Computing**: With the increasing adoption of edge computing, low latency is critical for processing data closer to the user. This reduces latency even further, enabling real-time processing and analysis of data.\\n9. **Real-time Analytics**: Low latency enables real-time analytics and insights, which are essential for applications like sentiment analysis, trend detection, and anomaly detection.\\n10. **Future-Proofing**: As LLMs continue to evolve and become more complex, low latency will become even more critical. By prioritizing low latency now, you'll be better prepared to handle the demands of future LLM applications.\\n\\nIn summary, low latency is vital for LLMs because it ensures a seamless user experience, improves model performance, and enables efficient resource utilization. By prioritizing low latency, you can build more effective, scalable, and efficient LLM applications that meet the demands of real-time interaction and processing.\", response_metadata={'token_usage': {'completion_tokens': 541, 'prompt_tokens': 33, 'total_tokens': 574, 'completion_time': 1.499777658, 'prompt_time': 0.008344704, 'queue_time': None, 'total_time': 1.508122362}, 'model_name': 'llama3-70b-8192', 'system_fingerprint': 'fp_87cbfbbc4d', 'finish_reason': 'stop', 'logprobs': None}, id='run-49dad960-ace8-4cd7-90b3-2db99ecbfa44-0')"
+      ]
+     },
+     "execution_count": 8,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
   "source": [
    "from langchain_core.prompts import ChatPromptTemplate\n",
-    "from langchain_groq import ChatGroq"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 28,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "chat = ChatGroq(temperature=0, model_name=\"mixtral-8x7b-32768\")"
+    "from langchain_groq import ChatGroq\n",
+    "\n",
+    "chat = ChatGroq(\n",
+    "    temperature=0,\n",
+    "    model=\"llama3-70b-8192\",\n",
+    "    # api_key=\"\" # Optional if not set as an environment variable\n",
+    ")\n",
+    "\n",
+    "system = \"You are a helpful assistant.\"\n",
+    "human = \"{text}\"\n",
+    "prompt = ChatPromptTemplate.from_messages([(\"system\", system), (\"human\", human)])\n",
+    "\n",
+    "chain = prompt | chat\n",
+    "chain.invoke({\"text\": \"Explain the importance of low latency for LLMs.\"})"
   ]
  },
  {
@@ -62,97 +89,206 @@
   "source": [
    "You can view the available models [here](https://console.groq.com/docs/models).\n",
    "\n",
-    "If you do not want to set your API key in the environment, you can pass it directly to the client:\n",
-    "```python\n",
-    "chat = ChatGroq(temperature=0, groq_api_key=\"YOUR_API_KEY\", model_name=\"mixtral-8x7b-32768\")\n",
+    "## Tool calling\n",
    "\n",
-    "```"
+    "Groq chat models support [tool calling](/docs/how_to/tool_calling/) to generate output matching a specific schema. The model may choose to call multiple tools or the same tool multiple times if appropriate.\n",
+    "\n",
+    "Here's an example:"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 10,
+   "metadata": {},
+   "outputs": [
+    {
+     "data": {
+      "text/plain": [
+       "[{'name': 'get_current_weather',\n",
+       "  'args': {'location': 'San Francisco', 'unit': 'Celsius'},\n",
+       "  'id': 'call_pydj'},\n",
+       " {'name': 'get_current_weather',\n",
+       "  'args': {'location': 'Tokyo', 'unit': 'Celsius'},\n",
+       "  'id': 'call_jgq3'}]"
+      ]
+     },
+     "execution_count": 10,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "from typing import Optional\n",
+    "\n",
+    "from langchain_core.tools import tool\n",
+    "\n",
+    "\n",
+    "@tool\n",
+    "def get_current_weather(location: str, unit: Optional[str]):\n",
+    "    \"\"\"Get the current weather in a given location\"\"\"\n",
+    "    return \"Cloudy with a chance of rain.\"\n",
+    "\n",
+    "\n",
+    "tool_model = chat.bind_tools([get_current_weather], tool_choice=\"auto\")\n",
+    "\n",
+    "res = tool_model.invoke(\"What is the weather like in San Francisco and Tokyo?\")\n",
+    "\n",
+    "res.tool_calls"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
-    "Write a prompt and invoke ChatGroq to create completions:"
+    "### `.with_structured_output()`\n",
+    "\n",
+    "You can also use the convenience [`.with_structured_output()`](/docs/how_to/structured_output/#the-with_structured_output-method) method to coerce `ChatGroq` into returning a structured output.\n",
+    "Here is an example:"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 29,
+   "execution_count": 11,
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
-       "AIMessage(content='Low Latency Large Language Models (LLMs) are a type of artificial intelligence model that can understand and generate human-like text. The term \"low latency\" refers to the model\\'s ability to process and respond to inputs quickly, with minimal delay.\\n\\nThe importance of low latency in LLMs can be explained through the following points:\\n\\n1. Improved user experience: In real-time applications such as chatbots, virtual assistants, and interactive games, users expect quick and responsive interactions. Low latency LLMs can provide instant feedback and responses, creating a more seamless and engaging user experience.\\n\\n2. Better decision-making: In time-sensitive scenarios, such as financial trading or autonomous vehicles, low latency LLMs can quickly process and analyze vast amounts of data, enabling faster and more informed decision-making.\\n\\n3. Enhanced accessibility: For individuals with disabilities, low latency LLMs can help create more responsive and inclusive interfaces, such as voice-controlled assistants or real-time captioning systems.\\n\\n4. Competitive advantage: In industries where real-time data analysis and decision-making are crucial, low latency LLMs can provide a competitive edge by enabling businesses to react more quickly to market changes, customer needs, or emerging opportunities.\\n\\n5. Scalability: Low latency LLMs can efficiently handle a higher volume of requests and interactions, making them more suitable for large-scale applications and services.\\n\\nIn summary, low latency is an essential aspect of LLMs, as it significantly impacts user experience, decision-making, accessibility, competitiveness, and scalability. By minimizing delays and response times, low latency LLMs can unlock new possibilities and applications for artificial intelligence in various industries and scenarios.')"
+       "Joke(setup='Why did the cat join a band?', punchline='Because it wanted to be the purr-cussionist!', rating=None)"
      ]
     },
-     "execution_count": 29,
+     "execution_count": 11,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
-    "system = \"You are a helpful assistant.\"\n",
-    "human = \"{text}\"\n",
-    "prompt = ChatPromptTemplate.from_messages([(\"system\", system), (\"human\", human)])\n",
+    "from langchain_core.pydantic_v1 import BaseModel, Field\n",
    "\n",
-    "chain = prompt | chat\n",
-    "chain.invoke({\"text\": \"Explain the importance of low latency LLMs.\"})"
+    "\n",
+    "class Joke(BaseModel):\n",
+    "    \"\"\"Joke to tell user.\"\"\"\n",
+    "\n",
+    "    setup: str = Field(description=\"The setup of the joke\")\n",
+    "    punchline: str = Field(description=\"The punchline to the joke\")\n",
+    "    rating: Optional[int] = Field(description=\"How funny the joke is, from 1 to 10\")\n",
+    "\n",
+    "\n",
+    "structured_llm = chat.with_structured_output(Joke)\n",
+    "\n",
+    "structured_llm.invoke(\"Tell me a joke about cats\")"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
-    "## `ChatGroq` also supports async and streaming functionality:"
+    "Behind the scenes, this takes advantage of the above tool calling functionality.\n",
+    "\n",
+    "## Async"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 32,
+   "execution_count": 12,
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
-       "AIMessage(content=\"There's a star that shines up in the sky,\\nThe Sun, that makes the day bright and spry.\\nIt rises and sets,\\nIn a daily, predictable bet,\\nGiving life to the world, oh my!\")"
+       "AIMessage(content='Here is a limerick about the sun:\\n\\nThere once was a sun in the sky,\\nWhose warmth and light caught the eye,\\nIt shone bright and bold,\\nWith a fiery gold,\\nAnd brought life to all, as it flew by.', response_metadata={'token_usage': {'completion_tokens': 51, 'prompt_tokens': 18, 'total_tokens': 69, 'completion_time': 0.144614022, 'prompt_time': 0.00585394, 'queue_time': None, 'total_time': 0.150467962}, 'model_name': 'llama3-70b-8192', 'system_fingerprint': 'fp_2f30b0b571', 'finish_reason': 'stop', 'logprobs': None}, id='run-e42340ba-f0ad-4b54-af61-8308d8ec8256-0')"
      ]
     },
-     "execution_count": 32,
+     "execution_count": 12,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
-    "chat = ChatGroq(temperature=0, model_name=\"mixtral-8x7b-32768\")\n",
+    "chat = ChatGroq(temperature=0, model=\"llama3-70b-8192\")\n",
    "prompt = ChatPromptTemplate.from_messages([(\"human\", \"Write a Limerick about {topic}\")])\n",
    "chain = prompt | chat\n",
    "await chain.ainvoke({\"topic\": \"The Sun\"})"
   ]
  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "## Streaming"
+   ]
+  },
  {
   "cell_type": "code",
-   "execution_count": 33,
+   "execution_count": 13,
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
-      "The moon's gentle glow\n",
-      "Illuminates the night sky\n",
-      "Peaceful and serene"
+      "Silvery glow bright\n",
+      "Luna's gentle light shines down\n",
+      "Midnight's gentle queen"
     ]
    }
   ],
   "source": [
-    "chat = ChatGroq(temperature=0, model_name=\"llama2-70b-4096\")\n",
+    "chat = ChatGroq(temperature=0, model=\"llama3-70b-8192\")\n",
    "prompt = ChatPromptTemplate.from_messages([(\"human\", \"Write a haiku about {topic}\")])\n",
    "chain = prompt | chat\n",
    "for chunk in chain.stream({\"topic\": \"The Moon\"}):\n",
    "    print(chunk.content, end=\"\", flush=True)"
   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "## Passing custom parameters\n",
+    "\n",
+    "You can pass other Groq-specific parameters using the `model_kwargs` argument on initialization. Here's an example of enabling JSON mode:"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 15,
+   "metadata": {},
+   "outputs": [
+    {
+     "data": {
+      "text/plain": [
+       "AIMessage(content='{ \"response\": \"That\\'s a tough question! There are eight species of bears found in the world, and each one is unique and amazing in its own way. However, if I had to pick one, I\\'d say the giant panda is a popular favorite among many people. Who can resist those adorable black and white markings?\", \"followup_question\": \"Would you like to know more about the giant panda\\'s habitat and diet?\" }', response_metadata={'token_usage': {'completion_tokens': 89, 'prompt_tokens': 50, 'total_tokens': 139, 'completion_time': 0.249032839, 'prompt_time': 0.011134497, 'queue_time': None, 'total_time': 0.260167336}, 'model_name': 'llama3-70b-8192', 'system_fingerprint': 'fp_2f30b0b571', 'finish_reason': 'stop', 'logprobs': None}, id='run-558ce67e-8c63-43fe-a48f-6ecf181bc922-0')"
+      ]
+     },
+     "execution_count": 15,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "chat = ChatGroq(\n",
+    "    model=\"llama3-70b-8192\", model_kwargs={\"response_format\": {\"type\": \"json_object\"}}\n",
+    ")\n",
+    "\n",
+    "system = \"\"\"\n",
+    "You are a helpful assistant.\n",
+    "Always respond with a JSON object with two string keys: \"response\" and \"followup_question\".\n",
+    "\"\"\"\n",
+    "human = \"{question}\"\n",
+    "prompt = ChatPromptTemplate.from_messages([(\"system\", system), (\"human\", human)])\n",
+    "\n",
+    "chain = prompt | chat\n",
+    "\n",
+    "chain.invoke({\"question\": \"what bear is best?\"})"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": []
  }
 ],
 "metadata": {
@@ -171,7 +307,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
-   "version": "3.10.13"
+   "version": "3.10.5"
  }
 },
 "nbformat": 4,
--- a/docs/docs/integrations/chat/mistralai.ipynb
+++ b/docs/docs/integrations/chat/mistralai.ipynb
@@ -225,7 +225,7 @@
   "source": [
    "## Chaining\n",
    "\n",
-    "You can also easily combine with a prompt template for easy structuring of user input. We can do this using [LCEL](/docs/concepts#langchain-expression-language)"
+    "You can also easily combine with a prompt template for easy structuring of user input. We can do this using [LCEL](/docs/concepts#langchain-expression-language-lcel)"
   ]
  },
  {
--- a/docs/docs/integrations/document_loaders/apify_dataset.ipynb
+++ b/docs/docs/integrations/document_loaders/apify_dataset.ipynb
@@ -13,7 +13,7 @@
    "\n",
    "## Prerequisites\n",
    "\n",
-    "You need to have an existing dataset on the Apify platform. If you don't have one, please first check out [this notebook](/docs/integrations/tools/apify) on how to use Apify to extract content from documentation, knowledge bases, help centers, or blogs."
+    "You need to have an existing dataset on the Apify platform. If you don't have one, please first check out [this notebook](/docs/integrations/tools/apify) on how to use Apify to extract content from documentation, knowledge bases, help centers, or blogs. This example shows how to load a dataset produced by the [Website Content Crawler](https://apify.com/apify/website-content-crawler)."
   ]
  },
  {
@@ -101,8 +101,10 @@
   "outputs": [],
   "source": [
    "from langchain.indexes import VectorstoreIndexCreator\n",
-    "from langchain_community.document_loaders import ApifyDatasetLoader\n",
-    "from langchain_core.documents import Document"
+    "from langchain_community.utilities import ApifyWrapper\n",
+    "from langchain_core.documents import Document\n",
+    "from langchain_openai import OpenAI\n",
+    "from langchain_openai.embeddings import OpenAIEmbeddings"
   ]
  },
  {
@@ -125,7 +127,7 @@
   "metadata": {},
   "outputs": [],
   "source": [
-    "index = VectorstoreIndexCreator().from_loaders([loader])"
+    "index = VectorstoreIndexCreator(embedding=OpenAIEmbeddings()).from_loaders([loader])"
   ]
  },
  {
@@ -135,7 +137,7 @@
   "outputs": [],
   "source": [
    "query = \"What is Apify?\"\n",
-    "result = index.query_with_sources(query)"
+    "result = index.query_with_sources(query, llm=OpenAI())"
   ]
  },
  {
--- a/docs/docs/integrations/document_loaders/async_chromium.ipynb
+++ b/docs/docs/integrations/document_loaders/async_chromium.ipynb
@@ -48,7 +48,7 @@
    "from langchain_community.document_loaders import AsyncChromiumLoader\n",
    "\n",
    "urls = [\"https://www.wsj.com\"]\n",
-    "loader = AsyncChromiumLoader(urls)\n",
+    "loader = AsyncChromiumLoader(urls, user_agent=\"MyAppUserAgent\")\n",
    "docs = loader.load()\n",
    "docs[0].page_content[0:100]"
   ]
--- a/docs/docs/integrations/document_loaders/kinetica.ipynb
+++ b/docs/docs/integrations/document_loaders/kinetica.ipynb
@@ -15,7 +15,7 @@
   "metadata": {},
   "outputs": [],
   "source": [
-    "%pip install gpudb==7.2.0.1"
+    "%pip install gpudb==7.2.0.9"
   ]
  },
  {
@@ -97,14 +97,14 @@
    "# data and the `SCHEMA.TABLE` combination must exist in Kinetica.\n",
    "\n",
    "QUERY = \"select text, survey_id as source from SCHEMA.TABLE limit 10\"\n",
-    "snowflake_loader = KineticaLoader(\n",
+    "kl = KineticaLoader(\n",
    "    query=QUERY,\n",
    "    host=HOST,\n",
    "    username=USERNAME,\n",
    "    password=PASSWORD,\n",
    "    metadata_columns=[\"source\"],\n",
    ")\n",
-    "kinetica_documents = snowflake_loader.load()\n",
+    "kinetica_documents = kl.load()\n",
    "print(kinetica_documents)"
   ]
  }
--- a/docs/docs/integrations/document_loaders/recursive_url.ipynb
+++ b/docs/docs/integrations/document_loaders/recursive_url.ipynb
@@ -7,140 +7,99 @@
   "source": [
    "# Recursive URL\n",
    "\n",
-    "We may want to process load all URLs under a root directory.\n",
-    "\n",
-    "For example, let's look at the [Python 3.9 Document](https://docs.python.org/3.9/).\n",
-    "\n",
-    "This has many interesting child pages that we may want to read in bulk.\n",
-    "\n",
-    "Of course, the `WebBaseLoader` can load a list of pages. \n",
-    "\n",
-    "But, the challenge is traversing the tree of child pages and actually assembling that list!\n",
-    " \n",
-    "We do this using the `RecursiveUrlLoader`.\n",
-    "\n",
-    "This also gives us the flexibility to exclude some children, customize the extractor, and more."
+    "The `RecursiveUrlLoader` lets you recursively scrape all child links from a root URL and parse them into Documents."
   ]
  },
  {
   "cell_type": "markdown",
-   "id": "1be8094f",
+   "id": "947d29e7-3679-483d-973f-79ea3403a370",
   "metadata": {},
   "source": [
-    "# Parameters\n",
-    "- url: str, the target url to crawl.\n",
-    "- exclude_dirs: Optional[str], webpage directories to exclude.\n",
-    "- use_async: Optional[bool], wether to use async requests, using async requests is usually faster in large tasks. However, async will disable the lazy loading feature(the function still works, but it is not lazy). By default, it is set to False.\n",
-    "- extractor: Optional[Callable[[str], str]], a function to extract the text of the document from the webpage, by default it returns the page as it is. It is recommended to use tools like goose3 and beautifulsoup to extract the text. By default, it just returns the page as it is.\n",
-    "- max_depth: Optional[int] = None, the maximum depth to crawl. By default, it is set to 2. If you need to crawl the whole website, set it to a number that is large enough would simply do the job.\n",
-    "- timeout: Optional[int] = None, the timeout for each request, in the unit of seconds. By default, it is set to 10.\n",
-    "- prevent_outside: Optional[bool] = None, whether to prevent crawling outside the root url. By default, it is set to True."
+    "## Setup\n",
+    "\n",
+    "The `RecursiveUrlLoader` lives in the `langchain-community` package. There's no other required packages, though you will get richer default Document metadata if you have ``beautifulsoup4` installed as well."
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
-   "id": "23c18539",
+   "id": "23359ab0-8056-4dee-8bff-c38dc079f17f",
   "metadata": {},
   "outputs": [],
   "source": [
-    "from langchain_community.document_loaders.recursive_url_loader import RecursiveUrlLoader"
+    "%pip install -qU langchain-community beautifulsoup4"
   ]
  },
  {
   "cell_type": "markdown",
-   "id": "6384c057",
+   "id": "07985766-e4e9-4ea1-8a18-924fa4f294e5",
   "metadata": {},
   "source": [
-    "Let's try a simple example."
+    "## Instantiation\n",
+    "\n",
+    "Now we can instantiate our document loader object and load Documents:"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": null,
-   "id": "55394afe",
+   "execution_count": 1,
+   "id": "cb208dcf-9ce9-4197-bc44-b80d20aa4e50",
   "metadata": {},
   "outputs": [],
   "source": [
-    "from bs4 import BeautifulSoup as Soup\n",
+    "from langchain_community.document_loaders import RecursiveUrlLoader\n",
    "\n",
-    "url = \"https://docs.python.org/3.9/\"\n",
    "loader = RecursiveUrlLoader(\n",
-    "    url=url, max_depth=2, extractor=lambda x: Soup(x, \"html.parser\").text\n",
-    ")\n",
-    "docs = loader.load()"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 3,
-   "id": "084fb2ce",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "'\\n\\n\\n\\n\\nPython Frequently Asked Questions — Python 3.'"
-      ]
-     },
-     "execution_count": 3,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "docs[0].page_content[:50]"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 4,
-   "id": "13bd7e16",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "{'source': 'https://docs.python.org/3.9/library/index.html',\n",
-       " 'title': 'The Python Standard Library — Python 3.9.17 documentation',\n",
-       " 'language': None}"
-      ]
-     },
-     "execution_count": 4,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "docs[-1].metadata"
+    "    \"https://docs.python.org/3.9/\",\n",
+    "    # max_depth=2,\n",
+    "    # use_async=False,\n",
+    "    # extractor=None,\n",
+    "    # metadata_extractor=None,\n",
+    "    # exclude_dirs=(),\n",
+    "    # timeout=10,\n",
+    "    # check_response_status=True,\n",
+    "    # continue_on_failure=True,\n",
+    "    # prevent_outside=True,\n",
+    "    # base_url=None,\n",
+    "    # ...\n",
+    ")"
   ]
  },
  {
   "cell_type": "markdown",
-   "id": "5866e5a6",
+   "id": "0fac4425-735f-487d-a12b-c8ed2a209039",
   "metadata": {},
   "source": [
-    "However, since it's hard to perform a perfect filter, you may still see some irrelevant results in the results. You can perform a filter on the returned documents by yourself, if it's needed. Most of the time, the returned results are good enough."
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "4ec8ecef",
-   "metadata": {},
-   "source": [
-    "Testing on LangChain docs."
+    "## Load\n",
+    "\n",
+    "Use ``.load()`` to synchronously load into memory all Documents, with one\n",
+    "Document per visited URL. Starting from the initial URL, we recurse through\n",
+    "all linked URLs up to the specified max_depth.\n",
+    "\n",
+    "Let's run through a basic example of how to use the `RecursiveUrlLoader` on the [Python 3.9 Documentation](https://docs.python.org/3.9/)."
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 2,
-   "id": "349b5598",
+   "id": "a30843c8-4a59-43dc-bf60-f26532f0f8e1",
   "metadata": {},
   "outputs": [
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "/Users/bagatur/.pyenv/versions/3.9.1/lib/python3.9/html/parser.py:170: XMLParsedAsHTMLWarning: It looks like you're parsing an XML document using an HTML parser. If this really is an HTML document (maybe it's XHTML?), you can ignore or filter this warning. If it's XML, you should know that using an XML parser will be more reliable. To parse this document as XML, make sure you have the lxml package installed, and pass the keyword argument `features=\"xml\"` into the BeautifulSoup constructor.\n",
+      "  k = self.parse_starttag(i)\n"
+     ]
+    },
    {
     "data": {
      "text/plain": [
-       "8"
+       "{'source': 'https://docs.python.org/3.9/',\n",
+       " 'content_type': 'text/html',\n",
+       " 'title': '3.9.19 Documentation',\n",
+       " 'language': None}"
      ]
     },
     "execution_count": 2,
@@ -149,10 +108,208 @@
    }
   ],
   "source": [
-    "url = \"https://js.langchain.com/docs/modules/memory/integrations/\"\n",
-    "loader = RecursiveUrlLoader(url=url)\n",
    "docs = loader.load()\n",
-    "len(docs)"
+    "docs[0].metadata"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "211856ed-6dd7-46c6-859e-11aaea9093db",
+   "metadata": {},
+   "source": [
+    "Great! The first document looks like the root page we started from. Let's look at the metadata of the next document"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 3,
+   "id": "2d842c03-fab8-4097-9f4f-809b2e71c0ba",
+   "metadata": {},
+   "outputs": [
+    {
+     "data": {
+      "text/plain": [
+       "{'source': 'https://docs.python.org/3.9/using/index.html',\n",
+       " 'content_type': 'text/html',\n",
+       " 'title': 'Python Setup and Usage — Python 3.9.19 documentation',\n",
+       " 'language': None}"
+      ]
+     },
+     "execution_count": 3,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "docs[1].metadata"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "f5714ace-7cc5-4c5c-9426-f68342880da0",
+   "metadata": {},
+   "source": [
+    "That url looks like a child of our root page, which is great! Let's move on from metadata to examine the content of one of our documents"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 6,
+   "id": "51dc6c67-6857-4298-9472-08b147f3a631",
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "\n",
+      "<!DOCTYPE html>\n",
+      "\n",
+      "<html xmlns=\"http://www.w3.org/1999/xhtml\">\n",
+      "  <head>\n",
+      "    <meta charset=\"utf-8\" /><title>3.9.19 Documentation</title><meta name=\"viewport\" content=\"width=device-width, initial-scale=1.0\">\n",
+      "    \n",
+      "    <link rel=\"stylesheet\" href=\"_static/pydoctheme.css\" type=\"text/css\" />\n",
+      "    <link rel=\n"
+     ]
+    }
+   ],
+   "source": [
+    "print(docs[0].page_content[:300])"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "d87cc239",
+   "metadata": {},
+   "source": [
+    "That certainly looks like HTML that comes from the url https://docs.python.org/3.9/, which is what we expected. Let's now look at some variations we can make to our basic example that can be helpful in different situations. "
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "8f41cc89",
+   "metadata": {},
+   "source": [
+    "## Adding an Extractor\n",
+    "\n",
+    "By default the loader sets the raw HTML from each link as the Document page content. To parse this HTML into a more human/LLM-friendly format you can pass in a custom ``extractor`` method:"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 21,
+   "id": "33a6f5b8",
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "/var/folders/td/vzm913rx77x21csd90g63_7c0000gn/T/ipykernel_10935/1083427287.py:6: XMLParsedAsHTMLWarning: It looks like you're parsing an XML document using an HTML parser. If this really is an HTML document (maybe it's XHTML?), you can ignore or filter this warning. If it's XML, you should know that using an XML parser will be more reliable. To parse this document as XML, make sure you have the lxml package installed, and pass the keyword argument `features=\"xml\"` into the BeautifulSoup constructor.\n",
+      "  soup = BeautifulSoup(html, \"lxml\")\n",
+      "/Users/isaachershenson/.pyenv/versions/3.11.9/lib/python3.11/html/parser.py:170: XMLParsedAsHTMLWarning: It looks like you're parsing an XML document using an HTML parser. If this really is an HTML document (maybe it's XHTML?), you can ignore or filter this warning. If it's XML, you should know that using an XML parser will be more reliable. To parse this document as XML, make sure you have the lxml package installed, and pass the keyword argument `features=\"xml\"` into the BeautifulSoup constructor.\n",
+      "  k = self.parse_starttag(i)\n"
+     ]
+    },
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "3.9.19 Documentation\n",
+      "\n",
+      "Download\n",
+      "Download these documents\n",
+      "Docs by version\n",
+      "\n",
+      "Python 3.13 (in development)\n",
+      "Python 3.12 (stable)\n",
+      "Python 3.11 (security-fixes)\n",
+      "Python 3.10 (security-fixes)\n",
+      "Python 3.9 (securit\n"
+     ]
+    }
+   ],
+   "source": [
+    "import re\n",
+    "\n",
+    "from bs4 import BeautifulSoup\n",
+    "\n",
+    "\n",
+    "def bs4_extractor(html: str) -> str:\n",
+    "    soup = BeautifulSoup(html, \"lxml\")\n",
+    "    return re.sub(r\"\\n\\n+\", \"\\n\\n\", soup.text).strip()\n",
+    "\n",
+    "\n",
+    "loader = RecursiveUrlLoader(\"https://docs.python.org/3.9/\", extractor=bs4_extractor)\n",
+    "docs = loader.load()\n",
+    "print(docs[0].page_content[:200])"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "c8e8a826",
+   "metadata": {},
+   "source": [
+    "This looks much nicer!\n",
+    "\n",
+    "You can similarly pass in a `metadata_extractor` to customize how Document metadata is extracted from the HTTP response. See the [API reference](https://api.python.langchain.com/en/latest/document_loaders/langchain_community.document_loaders.recursive_url_loader.RecursiveUrlLoader.html) for more on this."
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "1dddbc94",
+   "metadata": {},
+   "source": [
+    "## Lazy loading\n",
+    "\n",
+    "If we're loading a  large number of Documents and our downstream operations can be done over subsets of all loaded Documents, we can lazily load our Documents one at a time to minimize our memory footprint:"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 15,
+   "id": "7d0114fc",
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "/var/folders/4j/2rz3865x6qg07tx43146py8h0000gn/T/ipykernel_73962/2110507528.py:6: XMLParsedAsHTMLWarning: It looks like you're parsing an XML document using an HTML parser. If this really is an HTML document (maybe it's XHTML?), you can ignore or filter this warning. If it's XML, you should know that using an XML parser will be more reliable. To parse this document as XML, make sure you have the lxml package installed, and pass the keyword argument `features=\"xml\"` into the BeautifulSoup constructor.\n",
+      "  soup = BeautifulSoup(html, \"lxml\")\n"
+     ]
+    }
+   ],
+   "source": [
+    "page = []\n",
+    "for doc in loader.lazy_load():\n",
+    "    page.append(doc)\n",
+    "    if len(page) >= 10:\n",
+    "        # do some paged operation, e.g.\n",
+    "        # index.upsert(page)\n",
+    "\n",
+    "        page = []"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "f88a7c2f-35df-4c3a-b238-f91be2674b96",
+   "metadata": {},
+   "source": [
+    "In this example we never have more than 10 Documents loaded into memory at a time."
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "3e4d1c8f",
+   "metadata": {},
+   "source": [
+    "## API reference\n",
+    "\n",
+    "These examples show just a few of the ways in which you can modify the default `RecursiveUrlLoader`, but there are many more modifications that can be made to best fit your use case. Using the parameters `link_regex` and `exclude_dirs` can help you filter out unwanted URLs, `aload()` and `alazy_load()` can be used for aynchronous loading, and more.\n",
+    "\n",
+    "For detailed information on configuring and calling the ``RecursiveUrlLoader``, please see the API reference: https://api.python.langchain.com/en/latest/document_loaders/langchain_community.document_loaders.recursive_url_loader.RecursiveUrlLoader.html."
   ]
  }
 ],
@@ -172,7 +329,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
-   "version": "3.10.12"
+   "version": "3.11.9"
  }
 },
 "nbformat": 4,
--- a/docs/docs/integrations/document_loaders/source_code.ipynb
+++ b/docs/docs/integrations/document_loaders/source_code.ipynb
@@ -17,6 +17,7 @@
    "- C++ (*)\n",
    "- C# (*)\n",
    "- COBOL\n",
+    "- Elixir\n",
    "- Go (*)\n",
    "- Java (*)\n",
    "- JavaScript (requires package `esprima`)\n",
--- a/docs/docs/integrations/document_loaders/tomarkdown.ipynb
+++ b/docs/docs/integrations/document_loaders/tomarkdown.ipynb
@@ -113,7 +113,7 @@
      "\n",
      "LCEL is a declarative way to compose chains. LCEL was designed from day 1 to support putting prototypes in production, with no code changes, from the simplest “prompt + LLM” chain to the most complex chains.\n",
      "\n",
-      "- **[Overview](/docs/concepts#langchain-expression-language)**: LCEL and its benefits\n",
+      "- **[Overview](/docs/concepts#langchain-expression-language-lcel)**: LCEL and its benefits\n",
      "- **[Interface](/docs/concepts#interface)**: The standard interface for LCEL objects\n",
      "- **[How-to](/docs/expression_language/how_to)**: Key features of LCEL\n",
      "- **[Cookbook](/docs/expression_language/cookbook)**: Example code for accomplishing common tasks\n",
--- a/docs/docs/integrations/document_loaders/youtube_transcript.ipynb
+++ b/docs/docs/integrations/document_loaders/youtube_transcript.ipynb
@@ -15,47 +15,45 @@
  },
  {
   "cell_type": "code",
-   "execution_count": null,
   "id": "427d5745",
   "metadata": {},
+   "source": "from langchain_community.document_loaders import YoutubeLoader",
   "outputs": [],
-   "source": [
-    "from langchain_community.document_loaders import YoutubeLoader"
-   ]
+   "execution_count": null
  },
  {
   "cell_type": "code",
-   "execution_count": null,
   "id": "34a25b57",
   "metadata": {
    "scrolled": true
   },
-   "outputs": [],
   "source": [
    "%pip install --upgrade --quiet  youtube-transcript-api"
-   ]
+   ],
+   "outputs": [],
+   "execution_count": null
  },
  {
   "cell_type": "code",
-   "execution_count": null,
   "id": "bc8b308a",
   "metadata": {},
-   "outputs": [],
   "source": [
    "loader = YoutubeLoader.from_youtube_url(\n",
    "    \"https://www.youtube.com/watch?v=QsYGlZkevEg\", add_video_info=False\n",
    ")"
-   ]
+   ],
+   "outputs": [],
+   "execution_count": null
  },
  {
   "cell_type": "code",
-   "execution_count": null,
   "id": "d073dd36",
   "metadata": {},
-   "outputs": [],
   "source": [
    "loader.load()"
-   ]
+   ],
+   "outputs": [],
+   "execution_count": null
  },
  {
   "attachments": {},
@@ -68,26 +66,26 @@
  },
  {
   "cell_type": "code",
-   "execution_count": null,
   "id": "ba28af69",
   "metadata": {},
-   "outputs": [],
   "source": [
    "%pip install --upgrade --quiet  pytube"
-   ]
+   ],
+   "outputs": [],
+   "execution_count": null
  },
  {
   "cell_type": "code",
-   "execution_count": null,
   "id": "9b8ea390",
   "metadata": {},
-   "outputs": [],
   "source": [
    "loader = YoutubeLoader.from_youtube_url(\n",
    "    \"https://www.youtube.com/watch?v=QsYGlZkevEg\", add_video_info=True\n",
    ")\n",
    "loader.load()"
-   ]
+   ],
+   "outputs": [],
+   "execution_count": null
  },
  {
   "attachments": {},
@@ -104,10 +102,8 @@
  },
  {
   "cell_type": "code",
-   "execution_count": null,
   "id": "08510625",
   "metadata": {},
-   "outputs": [],
   "source": [
    "loader = YoutubeLoader.from_youtube_url(\n",
    "    \"https://www.youtube.com/watch?v=QsYGlZkevEg\",\n",
@@ -116,7 +112,41 @@
    "    translation=\"en\",\n",
    ")\n",
    "loader.load()"
-   ]
+   ],
+   "outputs": [],
+   "execution_count": null
+  },
+  {
+   "metadata": {},
+   "cell_type": "markdown",
+   "source": [
+    "### Get transcripts as timestamped chunks\n",
+    "\n",
+    "Get one or more `Document` objects, each containing a chunk of the video transcript.  The length of the chunks, in seconds, may be specified.  Each chunk's metadata includes a URL of the video on YouTube, which will start the video at the beginning of the specific chunk.\n",
+    "\n",
+    "`transcript_format` param:  One of the `langchain_community.document_loaders.youtube.TranscriptFormat` values.  In this case, `TranscriptFormat.CHUNKS`.\n",
+    "\n",
+    "`chunk_size_seconds` param:  An integer number of video seconds to be represented by each chunk of transcript data.  Default is 120 seconds."
+   ],
+   "id": "69f4e399a9764d73"
+  },
+  {
+   "metadata": {},
+   "cell_type": "code",
+   "source": [
+    "from langchain_community.document_loaders.youtube import TranscriptFormat\n",
+    "\n",
+    "loader = YoutubeLoader.from_youtube_url(\n",
+    "    \"https://www.youtube.com/watch?v=TKCMw0utiak\",\n",
+    "    add_video_info=True,\n",
+    "    transcript_format=TranscriptFormat.CHUNKS,\n",
+    "    chunk_size_seconds=30,\n",
+    ")\n",
+    "print(\"\\n\\n\".join(map(repr, loader.load())))"
+   ],
+   "id": "540bbf19182f38bc",
+   "outputs": [],
+   "execution_count": null
  },
  {
   "attachments": {},
@@ -142,10 +172,8 @@
  },
  {
   "cell_type": "code",
-   "execution_count": null,
   "id": "c345bc43",
   "metadata": {},
-   "outputs": [],
   "source": [
    "# Init the GoogleApiClient\n",
    "from pathlib import Path\n",
@@ -170,7 +198,9 @@
    "\n",
    "# returns a list of Documents\n",
    "youtube_loader_channel.load()"
-   ]
+   ],
+   "outputs": [],
+   "execution_count": null
  }
 ],
 "metadata": {
--- a/docs/docs/integrations/document_transformers/dashscope_rerank.ipynb
+++ b/docs/docs/integrations/document_transformers/dashscope_rerank.ipynb
@@ -0,0 +1,387 @@
+{
+ "cells": [
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "# DashScope Reranker\n",
+    "\n",
+    "This notebook shows how to use DashScope Reranker for document compression and retrieval. [DashScope](https://dashscope.aliyun.com/) is the generative AI service from Alibaba Cloud (Aliyun).\n",
+    "\n",
+    "DashScope's [Text ReRank Model](https://help.aliyun.com/document_detail/2780058.html?spm=a2c4g.2780059.0.0.6d995024FlrJ12) supports reranking documents with a maximum of 4000 tokens. Moreover, it supports Chinese, English, Japanese, Korean, Thai, Spanish, French, Portuguese, Indonesian, Arabic, and over 50 other languages. For more details, please visit [here](https://help.aliyun.com/document_detail/2780059.html?spm=a2c4g.2780058.0.0.3a9e5b1dWeOQjI)."
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "%pip install --upgrade --quiet  dashscope"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "%pip install --upgrade --quiet  faiss\n",
+    "\n",
+    "# OR  (depending on Python version)\n",
+    "\n",
+    "%pip install --upgrade --quiet  faiss-cpu"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "# To create api key: https://bailian.console.aliyun.com/?apiKey=1#/api-key\n",
+    "\n",
+    "import getpass\n",
+    "import os\n",
+    "\n",
+    "os.environ[\"DASHSCOPE_API_KEY\"] = getpass.getpass(\"DashScope API Key:\")"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 6,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "# Helper function for printing docs\n",
+    "def pretty_print_docs(docs):\n",
+    "    print(\n",
+    "        f\"\\n{'-' * 100}\\n\".join(\n",
+    "            [f\"Document {i+1}:\\n\\n\" + d.page_content for i, d in enumerate(docs)]\n",
+    "        )\n",
+    "    )"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "## Set up the base vector store retriever\n",
+    "Let's start by initializing a simple vector store retriever and storing the 2023 State of the Union speech (in chunks). We can set up the retriever to retrieve a high number (20) of docs."
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 7,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "Document 1:\n",
+      "\n",
+      "I understand. \n",
+      "\n",
+      "I remember when my Dad had to leave our home in Scranton, Pennsylvania to find work. I grew up in a family where if the price of food went up, you felt it. \n",
+      "\n",
+      "That’s why one of the first things I did as President was fight to pass the American Rescue Plan.  \n",
+      "\n",
+      "Because people were hurting. We needed to act, and we did. \n",
+      "\n",
+      "Few pieces of legislation have done more in a critical moment in our history to lift us out of crisis.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 2:\n",
+      "\n",
+      "One of the most serious constitutional responsibilities a President has is nominating someone to serve on the United States Supreme Court. \n",
+      "\n",
+      "And I did that 4 days ago, when I nominated Circuit Court of Appeals Judge Ketanji Brown Jackson. One of our nation’s top legal minds, who will continue Justice Breyer’s legacy of excellence.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 3:\n",
+      "\n",
+      "To all Americans, I will be honest with you, as I’ve always promised. A Russian dictator, invading a foreign country, has costs around the world. \n",
+      "\n",
+      "And I’m taking robust action to make sure the pain of our sanctions  is targeted at Russia’s economy. And I will use every tool at our disposal to protect American businesses and consumers. \n",
+      "\n",
+      "Tonight, I can announce that the United States has worked with 30 other countries to release 60 Million barrels of oil from reserves around the world.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 4:\n",
+      "\n",
+      "We cannot let this happen. \n",
+      "\n",
+      "Tonight. I call on the Senate to: Pass the Freedom to Vote Act. Pass the John Lewis Voting Rights Act. And while you’re at it, pass the Disclose Act so Americans can know who is funding our elections. \n",
+      "\n",
+      "Tonight, I’d like to honor someone who has dedicated his life to serve this country: Justice Stephen Breyer—an Army veteran, Constitutional scholar, and retiring Justice of the United States Supreme Court. Justice Breyer, thank you for your service.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 5:\n",
+      "\n",
+      "Tonight I say to the Russian oligarchs and corrupt leaders who have bilked billions of dollars off this violent regime no more. \n",
+      "\n",
+      "The U.S. Department of Justice is assembling a dedicated task force to go after the crimes of Russian oligarchs.  \n",
+      "\n",
+      "We are joining with our European allies to find and seize your yachts your luxury apartments your private jets. We are coming for your ill-begotten gains.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 6:\n",
+      "\n",
+      "Every Administration says they’ll do it, but we are actually doing it. \n",
+      "\n",
+      "We will buy American to make sure everything from the deck of an aircraft carrier to the steel on highway guardrails are made in America. \n",
+      "\n",
+      "But to compete for the best jobs of the future, we also need to level the playing field with China and other competitors.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 7:\n",
+      "\n",
+      "When we invest in our workers, when we build the economy from the bottom up and the middle out together, we can do something we haven’t done in a long time: build a better America. \n",
+      "\n",
+      "For more than two years, COVID-19 has impacted every decision in our lives and the life of the nation. \n",
+      "\n",
+      "And I know you’re tired, frustrated, and exhausted. \n",
+      "\n",
+      "But I also know this.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 8:\n",
+      "\n",
+      "A former top litigator in private practice. A former federal public defender. And from a family of public school educators and police officers. A consensus builder. Since she’s been nominated, she’s received a broad range of support—from the Fraternal Order of Police to former judges appointed by Democrats and Republicans. \n",
+      "\n",
+      "And if we are to advance liberty and justice, we need to secure the Border and fix the immigration system.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 9:\n",
+      "\n",
+      "My plan will not only lower costs to give families a fair shot, it will lower the deficit. \n",
+      "\n",
+      "The previous Administration not only ballooned the deficit with tax cuts for the very wealthy and corporations, it undermined the watchdogs whose job was to keep pandemic relief funds from being wasted. \n",
+      "\n",
+      "But in my administration, the watchdogs have been welcomed back. \n",
+      "\n",
+      "We’re going after the criminals who stole billions in relief money meant for small businesses and millions of Americans.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 10:\n",
+      "\n",
+      "He will never extinguish their love of freedom. He will never weaken the resolve of the free world. \n",
+      "\n",
+      "We meet tonight in an America that has lived through two of the hardest years this nation has ever faced. \n",
+      "\n",
+      "The pandemic has been punishing. \n",
+      "\n",
+      "And so many families are living paycheck to paycheck, struggling to keep up with the rising cost of food, gas, housing, and so much more. \n",
+      "\n",
+      "I understand.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 11:\n",
+      "\n",
+      "And tonight, I’m announcing that the Justice Department will name a chief prosecutor for pandemic fraud. \n",
+      "\n",
+      "By the end of this year, the deficit will be down to less than half what it was before I took office.  \n",
+      "\n",
+      "The only president ever to cut the deficit by more than one trillion dollars in a single year. \n",
+      "\n",
+      "Lowering your costs also means demanding more competition. \n",
+      "\n",
+      "I’m a capitalist, but capitalism without competition isn’t capitalism. \n",
+      "\n",
+      "It’s exploitation—and it drives up prices.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 12:\n",
+      "\n",
+      "Let each of us here tonight in this Chamber send an unmistakable signal to Ukraine and to the world. \n",
+      "\n",
+      "Please rise if you are able and show that, Yes, we the United States of America stand with the Ukrainian people. \n",
+      "\n",
+      "Throughout our history we’ve learned this lesson when dictators do not pay a price for their aggression they cause more chaos.   \n",
+      "\n",
+      "They keep moving.   \n",
+      "\n",
+      "And the costs and the threats to America and the world keep rising.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 13:\n",
+      "\n",
+      "Cancer is the #2 cause of death in America–second only to heart disease. \n",
+      "\n",
+      "Last month, I announced our plan to supercharge  \n",
+      "the Cancer Moonshot that President Obama asked me to lead six years ago. \n",
+      "\n",
+      "Our goal is to cut the cancer death rate by at least 50% over the next 25 years, turn more cancers from death sentences into treatable diseases.  \n",
+      "\n",
+      "More support for patients and families. \n",
+      "\n",
+      "To get there, I call on Congress to fund ARPA-H, the Advanced Research Projects Agency for Health.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 14:\n",
+      "\n",
+      "It fueled our efforts to vaccinate the nation and combat COVID-19. It delivered immediate economic relief for tens of millions of Americans.  \n",
+      "\n",
+      "Helped put food on their table, keep a roof over their heads, and cut the cost of health insurance. \n",
+      "\n",
+      "And as my Dad used to say, it gave people a little breathing room.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 15:\n",
+      "\n",
+      "America will lead that effort, releasing 30 Million barrels from our own Strategic Petroleum Reserve. And we stand ready to do more if necessary, unified with our allies.  \n",
+      "\n",
+      "These steps will help blunt gas prices here at home. And I know the news about what’s happening can seem alarming. \n",
+      "\n",
+      "But I want you to know that we are going to be okay. \n",
+      "\n",
+      "When the history of this era is written Putin’s war on Ukraine will have left Russia weaker and the rest of the world stronger.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 16:\n",
+      "\n",
+      "So that’s my plan. It will grow the economy and lower costs for families. \n",
+      "\n",
+      "So what are we waiting for? Let’s get this done. And while you’re at it, confirm my nominees to the Federal Reserve, which plays a critical role in fighting inflation.  \n",
+      "\n",
+      "My plan will not only lower costs to give families a fair shot, it will lower the deficit.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 17:\n",
+      "\n",
+      "And we will, as one people. \n",
+      "\n",
+      "One America. \n",
+      "\n",
+      "The United States of America. \n",
+      "\n",
+      "May God bless you all. May God protect our troops.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 18:\n",
+      "\n",
+      "As I’ve told Xi Jinping, it is never a good bet to bet against the American people. \n",
+      "\n",
+      "We’ll create good jobs for millions of Americans, modernizing roads, airports, ports, and waterways all across America. \n",
+      "\n",
+      "And we’ll do it all to withstand the devastating effects of the climate crisis and promote environmental justice.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 19:\n",
+      "\n",
+      "And I know you’re tired, frustrated, and exhausted. \n",
+      "\n",
+      "But I also know this. \n",
+      "\n",
+      "Because of the progress we’ve made, because of your resilience and the tools we have, tonight I can say  \n",
+      "we are moving forward safely, back to more normal routines.  \n",
+      "\n",
+      "We’ve reached a new moment in the fight against COVID-19, with severe cases down to a level not seen since last July.  \n",
+      "\n",
+      "Just a few days ago, the Centers for Disease Control and Prevention—the CDC—issued new mask guidelines.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 20:\n",
+      "\n",
+      "Madam Speaker, Madam Vice President, our First Lady and Second Gentleman. Members of Congress and the Cabinet. Justices of the Supreme Court. My fellow Americans.  \n",
+      "\n",
+      "Last year COVID-19 kept us apart. This year we are finally together again. \n",
+      "\n",
+      "Tonight, we meet as Democrats Republicans and Independents. But most importantly as Americans. \n",
+      "\n",
+      "With a duty to one another to the American people to the Constitution. \n",
+      "\n",
+      "And with an unwavering resolve that freedom will always triumph over tyranny.\n"
+     ]
+    }
+   ],
+   "source": [
+    "from langchain_community.document_loaders import TextLoader\n",
+    "from langchain_community.embeddings.dashscope import DashScopeEmbeddings\n",
+    "from langchain_community.vectorstores.faiss import FAISS\n",
+    "from langchain_text_splitters import RecursiveCharacterTextSplitter\n",
+    "\n",
+    "documents = TextLoader(\"../../how_to/state_of_the_union.txt\").load()\n",
+    "text_splitter = RecursiveCharacterTextSplitter(chunk_size=500, chunk_overlap=100)\n",
+    "texts = text_splitter.split_documents(documents)\n",
+    "retriever = FAISS.from_documents(texts, DashScopeEmbeddings()).as_retriever(  # type: ignore\n",
+    "    search_kwargs={\"k\": 20}\n",
+    ")\n",
+    "\n",
+    "query = \"What did the president say about Ketanji Brown Jackson\"\n",
+    "docs = retriever.invoke(query)\n",
+    "pretty_print_docs(docs)"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "## Reranking with DashScopeRerank\n",
+    "Now let's wrap our base retriever with a `ContextualCompressionRetriever`. We'll use the `DashScopeRerank` to rerank the returned results."
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 8,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "Document 1:\n",
+      "\n",
+      "One of the most serious constitutional responsibilities a President has is nominating someone to serve on the United States Supreme Court. \n",
+      "\n",
+      "And I did that 4 days ago, when I nominated Circuit Court of Appeals Judge Ketanji Brown Jackson. One of our nation’s top legal minds, who will continue Justice Breyer’s legacy of excellence.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 2:\n",
+      "\n",
+      "Madam Speaker, Madam Vice President, our First Lady and Second Gentleman. Members of Congress and the Cabinet. Justices of the Supreme Court. My fellow Americans.  \n",
+      "\n",
+      "Last year COVID-19 kept us apart. This year we are finally together again. \n",
+      "\n",
+      "Tonight, we meet as Democrats Republicans and Independents. But most importantly as Americans. \n",
+      "\n",
+      "With a duty to one another to the American people to the Constitution. \n",
+      "\n",
+      "And with an unwavering resolve that freedom will always triumph over tyranny.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 3:\n",
+      "\n",
+      "Tonight I say to the Russian oligarchs and corrupt leaders who have bilked billions of dollars off this violent regime no more. \n",
+      "\n",
+      "The U.S. Department of Justice is assembling a dedicated task force to go after the crimes of Russian oligarchs.  \n",
+      "\n",
+      "We are joining with our European allies to find and seize your yachts your luxury apartments your private jets. We are coming for your ill-begotten gains.\n"
+     ]
+    }
+   ],
+   "source": [
+    "from langchain.retrievers import ContextualCompressionRetriever\n",
+    "from langchain_community.document_compressors.dashscope_rerank import DashScopeRerank\n",
+    "\n",
+    "compressor = DashScopeRerank()\n",
+    "compression_retriever = ContextualCompressionRetriever(\n",
+    "    base_compressor=compressor, base_retriever=retriever\n",
+    ")\n",
+    "\n",
+    "compressed_docs = compression_retriever.invoke(\n",
+    "    \"What did the president say about Ketanji Jackson Brown\"\n",
+    ")\n",
+    "pretty_print_docs(compressed_docs)"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": []
+  }
+ ],
+ "metadata": {
+  "kernelspec": {
+   "display_name": "Python 3",
+   "language": "python",
+   "name": "python3"
+  },
+  "language_info": {
+   "codemirror_mode": {
+    "name": "ipython",
+    "version": 3
+   },
+   "file_extension": ".py",
+   "mimetype": "text/x-python",
+   "name": "python",
+   "nbconvert_exporter": "python",
+   "pygments_lexer": "ipython3",
+   "version": "3.10.13"
+  }
+ },
+ "nbformat": 4,
+ "nbformat_minor": 2
+}
--- a/docs/docs/integrations/document_transformers/volcengine_rerank.ipynb
+++ b/docs/docs/integrations/document_transformers/volcengine_rerank.ipynb
@@ -0,0 +1,420 @@
+{
+ "cells": [
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "# Volcengine Reranker\n",
+    "\n",
+    "This notebook shows how to use Volcengine Reranker for document compression and retrieval. [Volcengine](https://www.volcengine.com/) is a cloud service platform developed by ByteDance, the parent company of TikTok.\n",
+    "\n",
+    "Volcengine's Rerank Service supports reranking up to 50 documents with a maximum of 4000 tokens. For more, please visit [here](https://www.volcengine.com/docs/84313/1254474) and [here](https://www.volcengine.com/docs/84313/1254605)."
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "%pip install --upgrade --quiet  volcengine"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "%pip install --upgrade --quiet  faiss\n",
+    "\n",
+    "# OR  (depending on Python version)\n",
+    "\n",
+    "%pip install --upgrade --quiet  faiss-cpu"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "# To obtain ak/sk: https://www.volcengine.com/docs/84313/1254488\n",
+    "\n",
+    "import getpass\n",
+    "import os\n",
+    "\n",
+    "os.environ[\"VOLC_API_AK\"] = getpass.getpass(\"Volcengine API AK:\")\n",
+    "os.environ[\"VOLC_API_SK\"] = getpass.getpass(\"Volcengine API SK:\")"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 3,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "# Helper function for printing docs\n",
+    "def pretty_print_docs(docs):\n",
+    "    print(\n",
+    "        f\"\\n{'-' * 100}\\n\".join(\n",
+    "            [f\"Document {i+1}:\\n\\n\" + d.page_content for i, d in enumerate(docs)]\n",
+    "        )\n",
+    "    )"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "## Set up the base vector store retriever\n",
+    "Let's start by initializing a simple vector store retriever and storing the 2023 State of the Union speech (in chunks). We can set up the retriever to retrieve a high number (20) of docs."
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 4,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "/Users/terminator/Developer/langchain/.venv/lib/python3.11/site-packages/sentence_transformers/cross_encoder/CrossEncoder.py:11: TqdmExperimentalWarning: Using `tqdm.autonotebook.tqdm` in notebook mode. Use `tqdm.tqdm` instead to force console mode (e.g. in jupyter console)\n",
+      "  from tqdm.autonotebook import tqdm, trange\n",
+      "/Users/terminator/Developer/langchain/.venv/lib/python3.11/site-packages/huggingface_hub/file_download.py:1132: FutureWarning: `resume_download` is deprecated and will be removed in version 1.0.0. Downloads always resume when possible. If you want to force a new download, use `force_download=True`.\n",
+      "  warnings.warn(\n"
+     ]
+    },
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "Document 1:\n",
+      "\n",
+      "One of the most serious constitutional responsibilities a President has is nominating someone to serve on the United States Supreme Court. \n",
+      "\n",
+      "And I did that 4 days ago, when I nominated Circuit Court of Appeals Judge Ketanji Brown Jackson. One of our nation’s top legal minds, who will continue Justice Breyer’s legacy of excellence.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 2:\n",
+      "\n",
+      "We cannot let this happen. \n",
+      "\n",
+      "Tonight. I call on the Senate to: Pass the Freedom to Vote Act. Pass the John Lewis Voting Rights Act. And while you’re at it, pass the Disclose Act so Americans can know who is funding our elections. \n",
+      "\n",
+      "Tonight, I’d like to honor someone who has dedicated his life to serve this country: Justice Stephen Breyer—an Army veteran, Constitutional scholar, and retiring Justice of the United States Supreme Court. Justice Breyer, thank you for your service.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 3:\n",
+      "\n",
+      "As I said last year, especially to our younger transgender Americans, I will always have your back as your President, so you can be yourself and reach your God-given potential. \n",
+      "\n",
+      "While it often appears that we never agree, that isn’t true. I signed 80 bipartisan bills into law last year. From preventing government shutdowns to protecting Asian-Americans from still-too-common hate crimes to reforming military justice.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 4:\n",
+      "\n",
+      "He will never extinguish their love of freedom. He will never weaken the resolve of the free world. \n",
+      "\n",
+      "We meet tonight in an America that has lived through two of the hardest years this nation has ever faced. \n",
+      "\n",
+      "The pandemic has been punishing. \n",
+      "\n",
+      "And so many families are living paycheck to paycheck, struggling to keep up with the rising cost of food, gas, housing, and so much more. \n",
+      "\n",
+      "I understand.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 5:\n",
+      "\n",
+      "As Ohio Senator Sherrod Brown says, “It’s time to bury the label “Rust Belt.” \n",
+      "\n",
+      "It’s time. \n",
+      "\n",
+      "But with all the bright spots in our economy, record job growth and higher wages, too many families are struggling to keep up with the bills.  \n",
+      "\n",
+      "Inflation is robbing them of the gains they might otherwise feel. \n",
+      "\n",
+      "I get it. That’s why my top priority is getting prices under control.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 6:\n",
+      "\n",
+      "A former top litigator in private practice. A former federal public defender. And from a family of public school educators and police officers. A consensus builder. Since she’s been nominated, she’s received a broad range of support—from the Fraternal Order of Police to former judges appointed by Democrats and Republicans. \n",
+      "\n",
+      "And if we are to advance liberty and justice, we need to secure the Border and fix the immigration system.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 7:\n",
+      "\n",
+      "It’s not only the right thing to do—it’s the economically smart thing to do. \n",
+      "\n",
+      "That’s why immigration reform is supported by everyone from labor unions to religious leaders to the U.S. Chamber of Commerce. \n",
+      "\n",
+      "Let’s get it done once and for all. \n",
+      "\n",
+      "Advancing liberty and justice also requires protecting the rights of women. \n",
+      "\n",
+      "The constitutional right affirmed in Roe v. Wade—standing precedent for half a century—is under attack as never before.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 8:\n",
+      "\n",
+      "I understand. \n",
+      "\n",
+      "I remember when my Dad had to leave our home in Scranton, Pennsylvania to find work. I grew up in a family where if the price of food went up, you felt it. \n",
+      "\n",
+      "That’s why one of the first things I did as President was fight to pass the American Rescue Plan.  \n",
+      "\n",
+      "Because people were hurting. We needed to act, and we did. \n",
+      "\n",
+      "Few pieces of legislation have done more in a critical moment in our history to lift us out of crisis.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 9:\n",
+      "\n",
+      "Third – we can end the shutdown of schools and businesses. We have the tools we need. \n",
+      "\n",
+      "It’s time for Americans to get back to work and fill our great downtowns again.  People working from home can feel safe to begin to return to the office.   \n",
+      "\n",
+      "We’re doing that here in the federal government. The vast majority of federal workers will once again work in person. \n",
+      "\n",
+      "Our schools are open. Let’s keep it that way. Our kids need to be in school.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 10:\n",
+      "\n",
+      "He met the Ukrainian people. \n",
+      "\n",
+      "From President Zelenskyy to every Ukrainian, their fearlessness, their courage, their determination, inspires the world. \n",
+      "\n",
+      "Groups of citizens blocking tanks with their bodies. Everyone from students to retirees teachers turned soldiers defending their homeland. \n",
+      "\n",
+      "In this struggle as President Zelenskyy said in his speech to the European Parliament “Light will win over darkness.” The Ukrainian Ambassador to the United States is here tonight.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 11:\n",
+      "\n",
+      "The widow of Sergeant First Class Heath Robinson.  \n",
+      "\n",
+      "He was born a soldier. Army National Guard. Combat medic in Kosovo and Iraq. \n",
+      "\n",
+      "Stationed near Baghdad, just yards from burn pits the size of football fields. \n",
+      "\n",
+      "Heath’s widow Danielle is here with us tonight. They loved going to Ohio State football games. He loved building Legos with their daughter. \n",
+      "\n",
+      "But cancer from prolonged exposure to burn pits ravaged Heath’s lungs and body. \n",
+      "\n",
+      "Danielle says Heath was a fighter to the very end.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 12:\n",
+      "\n",
+      "Danielle says Heath was a fighter to the very end. \n",
+      "\n",
+      "He didn’t know how to stop fighting, and neither did she. \n",
+      "\n",
+      "Through her pain she found purpose to demand we do better. \n",
+      "\n",
+      "Tonight, Danielle—we are. \n",
+      "\n",
+      "The VA is pioneering new ways of linking toxic exposures to diseases, already helping more veterans get benefits. \n",
+      "\n",
+      "And tonight, I’m announcing we’re expanding eligibility to veterans suffering from nine respiratory cancers.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 13:\n",
+      "\n",
+      "We can do all this while keeping lit the torch of liberty that has led generations of immigrants to this land—my forefathers and so many of yours. \n",
+      "\n",
+      "Provide a pathway to citizenship for Dreamers, those on temporary status, farm workers, and essential workers. \n",
+      "\n",
+      "Revise our laws so businesses have the workers they need and families don’t wait decades to reunite. \n",
+      "\n",
+      "It’s not only the right thing to do—it’s the economically smart thing to do.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 14:\n",
+      "\n",
+      "He rejected repeated efforts at diplomacy. \n",
+      "\n",
+      "He thought the West and NATO wouldn’t respond. And he thought he could divide us at home. Putin was wrong. We were ready.  Here is what we did.   \n",
+      "\n",
+      "We prepared extensively and carefully. \n",
+      "\n",
+      "We spent months building a coalition of other freedom-loving nations from Europe and the Americas to Asia and Africa to confront Putin.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 15:\n",
+      "\n",
+      "As I’ve told Xi Jinping, it is never a good bet to bet against the American people. \n",
+      "\n",
+      "We’ll create good jobs for millions of Americans, modernizing roads, airports, ports, and waterways all across America. \n",
+      "\n",
+      "And we’ll do it all to withstand the devastating effects of the climate crisis and promote environmental justice.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 16:\n",
+      "\n",
+      "Tonight I say to the Russian oligarchs and corrupt leaders who have bilked billions of dollars off this violent regime no more. \n",
+      "\n",
+      "The U.S. Department of Justice is assembling a dedicated task force to go after the crimes of Russian oligarchs.  \n",
+      "\n",
+      "We are joining with our European allies to find and seize your yachts your luxury apartments your private jets. We are coming for your ill-begotten gains.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 17:\n",
+      "\n",
+      "Look at cars. \n",
+      "\n",
+      "Last year, there weren’t enough semiconductors to make all the cars that people wanted to buy. \n",
+      "\n",
+      "And guess what, prices of automobiles went up. \n",
+      "\n",
+      "So—we have a choice. \n",
+      "\n",
+      "One way to fight inflation is to drive down wages and make Americans poorer.  \n",
+      "\n",
+      "I have a better plan to fight inflation. \n",
+      "\n",
+      "Lower your costs, not your wages. \n",
+      "\n",
+      "Make more cars and semiconductors in America. \n",
+      "\n",
+      "More infrastructure and innovation in America. \n",
+      "\n",
+      "More goods moving faster and cheaper in America.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 18:\n",
+      "\n",
+      "So that’s my plan. It will grow the economy and lower costs for families. \n",
+      "\n",
+      "So what are we waiting for? Let’s get this done. And while you’re at it, confirm my nominees to the Federal Reserve, which plays a critical role in fighting inflation.  \n",
+      "\n",
+      "My plan will not only lower costs to give families a fair shot, it will lower the deficit.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 19:\n",
+      "\n",
+      "Let each of us here tonight in this Chamber send an unmistakable signal to Ukraine and to the world. \n",
+      "\n",
+      "Please rise if you are able and show that, Yes, we the United States of America stand with the Ukrainian people. \n",
+      "\n",
+      "Throughout our history we’ve learned this lesson when dictators do not pay a price for their aggression they cause more chaos.   \n",
+      "\n",
+      "They keep moving.   \n",
+      "\n",
+      "And the costs and the threats to America and the world keep rising.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 20:\n",
+      "\n",
+      "It’s based on DARPA—the Defense Department project that led to the Internet, GPS, and so much more.  \n",
+      "\n",
+      "ARPA-H will have a singular purpose—to drive breakthroughs in cancer, Alzheimer’s, diabetes, and more. \n",
+      "\n",
+      "A unity agenda for the nation. \n",
+      "\n",
+      "We can do this. \n",
+      "\n",
+      "My fellow Americans—tonight , we have gathered in a sacred space—the citadel of our democracy. \n",
+      "\n",
+      "In this Capitol, generation after generation, Americans have debated great questions amid great strife, and have done great things.\n"
+     ]
+    },
+    {
+     "name": "stderr",
+     "output_type": "stream",
+     "text": [
+      "huggingface/tokenizers: The current process just got forked, after parallelism has already been used. Disabling parallelism to avoid deadlocks...\n",
+      "To disable this warning, you can either:\n",
+      "\t- Avoid using `tokenizers` before the fork if possible\n",
+      "\t- Explicitly set the environment variable TOKENIZERS_PARALLELISM=(true | false)\n"
+     ]
+    }
+   ],
+   "source": [
+    "from langchain_community.document_loaders import TextLoader\n",
+    "from langchain_community.vectorstores.faiss import FAISS\n",
+    "from langchain_huggingface import HuggingFaceEmbeddings\n",
+    "from langchain_text_splitters import RecursiveCharacterTextSplitter\n",
+    "\n",
+    "documents = TextLoader(\"../../how_to/state_of_the_union.txt\").load()\n",
+    "text_splitter = RecursiveCharacterTextSplitter(chunk_size=500, chunk_overlap=100)\n",
+    "texts = text_splitter.split_documents(documents)\n",
+    "retriever = FAISS.from_documents(\n",
+    "    texts, HuggingFaceEmbeddings(model_name=\"all-MiniLM-L6-v2\")\n",
+    ").as_retriever(search_kwargs={\"k\": 20})\n",
+    "\n",
+    "query = \"What did the president say about Ketanji Brown Jackson\"\n",
+    "docs = retriever.invoke(query)\n",
+    "pretty_print_docs(docs)"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "## Reranking with VolcengineRerank\n",
+    "Now let's wrap our base retriever with a `ContextualCompressionRetriever`. We'll use the `VolcengineRerank` to rerank the returned results."
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 5,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "Document 1:\n",
+      "\n",
+      "One of the most serious constitutional responsibilities a President has is nominating someone to serve on the United States Supreme Court. \n",
+      "\n",
+      "And I did that 4 days ago, when I nominated Circuit Court of Appeals Judge Ketanji Brown Jackson. One of our nation’s top legal minds, who will continue Justice Breyer’s legacy of excellence.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 2:\n",
+      "\n",
+      "As I said last year, especially to our younger transgender Americans, I will always have your back as your President, so you can be yourself and reach your God-given potential. \n",
+      "\n",
+      "While it often appears that we never agree, that isn’t true. I signed 80 bipartisan bills into law last year. From preventing government shutdowns to protecting Asian-Americans from still-too-common hate crimes to reforming military justice.\n",
+      "----------------------------------------------------------------------------------------------------\n",
+      "Document 3:\n",
+      "\n",
+      "We cannot let this happen. \n",
+      "\n",
+      "Tonight. I call on the Senate to: Pass the Freedom to Vote Act. Pass the John Lewis Voting Rights Act. And while you’re at it, pass the Disclose Act so Americans can know who is funding our elections. \n",
+      "\n",
+      "Tonight, I’d like to honor someone who has dedicated his life to serve this country: Justice Stephen Breyer—an Army veteran, Constitutional scholar, and retiring Justice of the United States Supreme Court. Justice Breyer, thank you for your service.\n"
+     ]
+    }
+   ],
+   "source": [
+    "from langchain.retrievers import ContextualCompressionRetriever\n",
+    "from langchain_community.document_compressors.volcengine_rerank import VolcengineRerank\n",
+    "\n",
+    "compressor = VolcengineRerank()\n",
+    "compression_retriever = ContextualCompressionRetriever(\n",
+    "    base_compressor=compressor, base_retriever=retriever\n",
+    ")\n",
+    "\n",
+    "compressed_docs = compression_retriever.invoke(\n",
+    "    \"What did the president say about Ketanji Jackson Brown\"\n",
+    ")\n",
+    "pretty_print_docs(compressed_docs)"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": []
+  }
+ ],
+ "metadata": {
+  "kernelspec": {
+   "display_name": "Python 3",
+   "language": "python",
+   "name": "python3"
+  },
+  "language_info": {
+   "codemirror_mode": {
+    "name": "ipython",
+    "version": 3
+   },
+   "file_extension": ".py",
+   "mimetype": "text/x-python",
+   "name": "python",
+   "nbconvert_exporter": "python",
+   "pygments_lexer": "ipython3",
+   "version": "3.11.9"
+  }
+ },
+ "nbformat": 4,
+ "nbformat_minor": 2
+}
--- a/docs/docs/integrations/document_transformers/voyageai-reranker.ipynb
+++ b/docs/docs/integrations/document_transformers/voyageai-reranker.ipynb
@@ -90,7 +90,9 @@
    "- `voyage-code-2`\n",
    "- `voyage-2`\n",
    "- `voyage-law-2`\n",
-    "- `voyage-lite-02-instruct`"
+    "- `voyage-lite-02-instruct`\n",
+    "- `voyage-finance-2`\n",
+    "- `voyage-multilingual-2`"
   ]
  },
  {
@@ -336,7 +338,10 @@
   "metadata": {},
   "source": [
    "## Doing reranking with VoyageAIRerank\n",
-    "Now let's wrap our base retriever with a `ContextualCompressionRetriever`. We'll use the Voyage AI reranker to rerank the returned results."
+    "Now let's wrap our base retriever with a `ContextualCompressionRetriever`. We'll use the Voyage AI reranker to rerank the returned results. You can use any of the following Reranking models: ([source](https://docs.voyageai.com/docs/reranker)):\n",
+    "\n",
+    "- `rerank-1`\n",
+    "- `rerank-lite-1`"
   ]
  },
  {
--- a/docs/docs/integrations/graphs/diffbot.ipynb
+++ b/docs/docs/integrations/graphs/diffbot.ipynb
@@ -9,8 +9,7 @@
    "\n",
    ">[Diffbot](https://docs.diffbot.com/docs/getting-started-with-diffbot) is a suite of ML-based products that make it easy to structure web data.\n",
    ">\n",
-    ">Diffbot's [Natural Language Processing API](https://www.diffbot.com/products/natural-language/) allows for the extraction of entities, relationships, and semantic meaning from unstructured text data.",
-    "\n",
+    ">Diffbot's [Natural Language Processing API](https://www.diffbot.com/products/natural-language/) allows for the extraction of entities, relationships, and semantic meaning from unstructured text data.\n",
    "[![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/langchain-ai/langchain/blob/master/docs/docs/integrations/graphs/diffbot.ipynb)\n",
    "\n",
    "## Use case\n",
@@ -70,8 +69,8 @@
   "source": [
    "from langchain_experimental.graph_transformers.diffbot import DiffbotGraphTransformer\n",
    "\n",
-    "diffbot_api_token = \"DIFFBOT_API_TOKEN\"\n",
-    "diffbot_nlp = DiffbotGraphTransformer(diffbot_api_token=diffbot_api_token)"
+    "diffbot_api_key = \"DIFFBOT_KEY\"\n",
+    "diffbot_nlp = DiffbotGraphTransformer(diffbot_api_key=diffbot_api_key)"
   ]
  },
  {
@@ -111,7 +110,7 @@
    "    --name neo4j \\\n",
    "    -p 7474:7474 -p 7687:7687 \\\n",
    "    -d \\\n",
-    "    -e NEO4J_AUTH=neo4j/pleaseletmein \\\n",
+    "    -e NEO4J_AUTH=neo4j/password \\\n",
    "    -e NEO4J_PLUGINS=\\[\\\"apoc\\\"\\]  \\\n",
    "    neo4j:latest\n",
    "```    \n",
@@ -129,7 +128,7 @@
    "\n",
    "url = \"bolt://localhost:7687\"\n",
    "username = \"neo4j\"\n",
-    "password = \"pleaseletmein\"\n",
+    "password = \"password\"\n",
    "\n",
    "graph = Neo4jGraph(url=url, username=username, password=password)"
   ]
@@ -296,7 +295,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
-   "version": "3.10.12"
+   "version": "3.9.18"
  }
 },
 "nbformat": 4,
--- a/docs/docs/integrations/graphs/neo4j_cypher.ipynb
+++ b/docs/docs/integrations/graphs/neo4j_cypher.ipynb
@@ -164,10 +164,10 @@
     "text": [
      "Node properties:\n",
      "- **Movie**\n",
-      "  - `runtime: INTEGER` Min: 120, Max: 120\n",
-      "  - `name: STRING` Available options: ['Top Gun']\n",
+      "  - `runtime`: INTEGER Min: 120, Max: 120\n",
+      "  - `name`: STRING Available options: ['Top Gun']\n",
      "- **Actor**\n",
-      "  - `name: STRING` Available options: ['Tom Cruise', 'Val Kilmer', 'Anthony Edwards', 'Meg Ryan']\n",
+      "  - `name`: STRING Available options: ['Tom Cruise', 'Val Kilmer', 'Anthony Edwards', 'Meg Ryan']\n",
      "Relationship properties:\n",
      "\n",
      "The relationships:\n",
@@ -225,7 +225,7 @@
      "WHERE m.name = 'Top Gun'\n",
      "RETURN a.name\u001b[0m\n",
      "Full Context:\n",
-      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Tom Cruise'}]\u001b[0m\n",
+      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Tom Cruise'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}]\u001b[0m\n",
      "\n",
      "\u001b[1m> Finished chain.\u001b[0m\n"
     ]
@@ -234,7 +234,7 @@
     "data": {
      "text/plain": [
       "{'query': 'Who played in Top Gun?',\n",
-       " 'result': 'Anthony Edwards, Meg Ryan, Val Kilmer, Tom Cruise played in Top Gun.'}"
+       " 'result': 'Tom Cruise, Val Kilmer, Anthony Edwards, and Meg Ryan played in Top Gun.'}"
      ]
     },
     "execution_count": 8,
@@ -286,7 +286,7 @@
      "WHERE m.name = 'Top Gun'\n",
      "RETURN a.name\u001b[0m\n",
      "Full Context:\n",
-      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}]\u001b[0m\n",
+      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Tom Cruise'}, {'a.name': 'Val Kilmer'}]\u001b[0m\n",
      "\n",
      "\u001b[1m> Finished chain.\u001b[0m\n"
     ]
@@ -295,7 +295,7 @@
     "data": {
      "text/plain": [
       "{'query': 'Who played in Top Gun?',\n",
-       " 'result': 'Anthony Edwards, Meg Ryan played in Top Gun.'}"
+       " 'result': 'Tom Cruise, Val Kilmer played in Top Gun.'}"
      ]
     },
     "execution_count": 10,
@@ -346,11 +346,11 @@
      "WHERE m.name = 'Top Gun'\n",
      "RETURN a.name\u001b[0m\n",
      "Full Context:\n",
-      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Tom Cruise'}]\u001b[0m\n",
+      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Tom Cruise'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}]\u001b[0m\n",
      "\n",
      "\u001b[1m> Finished chain.\u001b[0m\n",
-      "Intermediate steps: [{'query': \"MATCH (a:Actor)-[:ACTED_IN]->(m:Movie)\\nWHERE m.name = 'Top Gun'\\nRETURN a.name\"}, {'context': [{'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Tom Cruise'}]}]\n",
-      "Final answer: Anthony Edwards, Meg Ryan, Val Kilmer, Tom Cruise played in Top Gun.\n"
+      "Intermediate steps: [{'query': \"MATCH (a:Actor)-[:ACTED_IN]->(m:Movie)\\nWHERE m.name = 'Top Gun'\\nRETURN a.name\"}, {'context': [{'a.name': 'Tom Cruise'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}]}]\n",
+      "Final answer: Tom Cruise, Val Kilmer, Anthony Edwards, and Meg Ryan played in Top Gun.\n"
     ]
    }
   ],
@@ -406,10 +406,10 @@
     "data": {
      "text/plain": [
       "{'query': 'Who played in Top Gun?',\n",
-       " 'result': [{'a.name': 'Anthony Edwards'},\n",
-       "  {'a.name': 'Meg Ryan'},\n",
+       " 'result': [{'a.name': 'Tom Cruise'},\n",
       "  {'a.name': 'Val Kilmer'},\n",
-       "  {'a.name': 'Tom Cruise'}]}"
+       "  {'a.name': 'Anthony Edwards'},\n",
+       "  {'a.name': 'Meg Ryan'}]}"
      ]
     },
     "execution_count": 14,
@@ -482,7 +482,7 @@
      "\n",
      "\u001b[1m> Entering new GraphCypherQAChain chain...\u001b[0m\n",
      "Generated Cypher:\n",
-      "\u001b[32;1m\u001b[1;3mMATCH (:Movie {name:\"Top Gun\"})<-[:ACTED_IN]-()\n",
+      "\u001b[32;1m\u001b[1;3mMATCH (m:Movie {name:\"Top Gun\"})<-[:ACTED_IN]-()\n",
      "RETURN count(*) AS numberOfActors\u001b[0m\n",
      "Full Context:\n",
      "\u001b[32;1m\u001b[1;3m[{'numberOfActors': 4}]\u001b[0m\n",
@@ -494,7 +494,7 @@
     "data": {
      "text/plain": [
       "{'query': 'How many people played in Top Gun?',\n",
-       " 'result': 'There were 4 actors who played in Top Gun.'}"
+       " 'result': 'There were 4 actors in Top Gun.'}"
      ]
     },
     "execution_count": 16,
@@ -548,7 +548,7 @@
      "WHERE m.name = 'Top Gun'\n",
      "RETURN a.name\u001b[0m\n",
      "Full Context:\n",
-      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Tom Cruise'}]\u001b[0m\n",
+      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Tom Cruise'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}]\u001b[0m\n",
      "\n",
      "\u001b[1m> Finished chain.\u001b[0m\n"
     ]
@@ -557,7 +557,7 @@
     "data": {
      "text/plain": [
       "{'query': 'Who played in Top Gun?',\n",
-       " 'result': 'Anthony Edwards, Meg Ryan, Val Kilmer, and Tom Cruise played in Top Gun.'}"
+       " 'result': 'Tom Cruise, Val Kilmer, Anthony Edwards, and Meg Ryan played in Top Gun.'}"
      ]
     },
     "execution_count": 18,
@@ -661,7 +661,7 @@
      "WHERE m.name = 'Top Gun'\n",
      "RETURN a.name\u001b[0m\n",
      "Full Context:\n",
-      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Tom Cruise'}]\u001b[0m\n",
+      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Tom Cruise'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}]\u001b[0m\n",
      "\n",
      "\u001b[1m> Finished chain.\u001b[0m\n"
     ]
@@ -670,7 +670,7 @@
     "data": {
      "text/plain": [
       "{'query': 'Who played in Top Gun?',\n",
-       " 'result': 'Anthony Edwards, Meg Ryan, Val Kilmer, Tom Cruise played in Top Gun.'}"
+       " 'result': 'Tom Cruise, Val Kilmer, Anthony Edwards, and Meg Ryan played in Top Gun.'}"
      ]
     },
     "execution_count": 22,
@@ -683,12 +683,116 @@
   ]
  },
  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "3fa3f3d5-f7e7-4ca9-8f07-ca22b897f192",
+   "cell_type": "markdown",
+   "id": "81093062-eb7f-4d96-b1fd-c36b8f1b9474",
   "metadata": {},
-   "outputs": [],
-   "source": []
+   "source": [
+    "## Provide context from database results as tool/function output\n",
+    "\n",
+    "You can use the `use_function_response` parameter to pass context from database results to an LLM as a tool/function output. This method improves the response accuracy and relevance of an answer as the LLM follows the provided context more closely.\n",
+    "_You will need to use an LLM with native function calling support to use this feature_."
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 23,
+   "id": "2be8f51c-e80a-4a60-ab1c-266450fc17cd",
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "\n",
+      "\n",
+      "\u001b[1m> Entering new GraphCypherQAChain chain...\u001b[0m\n",
+      "Generated Cypher:\n",
+      "\u001b[32;1m\u001b[1;3mMATCH (a:Actor)-[:ACTED_IN]->(m:Movie)\n",
+      "WHERE m.name = 'Top Gun'\n",
+      "RETURN a.name\u001b[0m\n",
+      "Full Context:\n",
+      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Tom Cruise'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}]\u001b[0m\n",
+      "\n",
+      "\u001b[1m> Finished chain.\u001b[0m\n"
+     ]
+    },
+    {
+     "data": {
+      "text/plain": [
+       "{'query': 'Who played in Top Gun?',\n",
+       " 'result': 'The main actors in Top Gun are Tom Cruise, Val Kilmer, Anthony Edwards, and Meg Ryan.'}"
+      ]
+     },
+     "execution_count": 23,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "chain = GraphCypherQAChain.from_llm(\n",
+    "    llm=ChatOpenAI(temperature=0, model=\"gpt-3.5-turbo\"),\n",
+    "    graph=graph,\n",
+    "    verbose=True,\n",
+    "    use_function_response=True,\n",
+    ")\n",
+    "chain.invoke({\"query\": \"Who played in Top Gun?\"})"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "48a75785-5bc9-49a7-a41b-88bf3ef9d312",
+   "metadata": {},
+   "source": [
+    "You can provide custom system message when using the function response feature by providing `function_response_system` to instruct the model on how to generate answers.\n",
+    "\n",
+    "_Note that `qa_prompt` will have no effect when using `use_function_response`_"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 24,
+   "id": "ddf0a61e-f104-4dbb-abbf-e65f3f57dd9a",
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "\n",
+      "\n",
+      "\u001b[1m> Entering new GraphCypherQAChain chain...\u001b[0m\n",
+      "Generated Cypher:\n",
+      "\u001b[32;1m\u001b[1;3mMATCH (a:Actor)-[:ACTED_IN]->(m:Movie)\n",
+      "WHERE m.name = 'Top Gun'\n",
+      "RETURN a.name\u001b[0m\n",
+      "Full Context:\n",
+      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Tom Cruise'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}]\u001b[0m\n",
+      "\n",
+      "\u001b[1m> Finished chain.\u001b[0m\n"
+     ]
+    },
+    {
+     "data": {
+      "text/plain": [
+       "{'query': 'Who played in Top Gun?',\n",
+       " 'result': \"Arrr matey! In the film Top Gun, ye be seein' Tom Cruise, Val Kilmer, Anthony Edwards, and Meg Ryan sailin' the high seas of the sky! Aye, they be a fine crew of actors, they be!\"}"
+      ]
+     },
+     "execution_count": 24,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "chain = GraphCypherQAChain.from_llm(\n",
+    "    llm=ChatOpenAI(temperature=0, model=\"gpt-3.5-turbo\"),\n",
+    "    graph=graph,\n",
+    "    verbose=True,\n",
+    "    use_function_response=True,\n",
+    "    function_response_system=\"Respond as a pirate!\",\n",
+    ")\n",
+    "chain.invoke({\"query\": \"Who played in Top Gun?\"})"
+   ]
  }
 ],
 "metadata": {
--- a/docs/docs/integrations/llm_caching.ipynb
+++ b/docs/docs/integrations/llm_caching.ipynb
@@ -724,6 +724,83 @@
    "llm(\"Tell me joke\")"
   ]
  },
+  {
+   "cell_type": "markdown",
+   "id": "9b2b2777",
+   "metadata": {},
+   "source": [
+    "## `MongoDB Atlas` Cache\n",
+    "\n",
+    "[MongoDB Atlas](https://www.mongodb.com/docs/atlas/) is a fully-managed cloud database available in AWS, Azure, and GCP. It has native support for \n",
+    "Vector Search on the MongoDB document data.\n",
+    "Use [MongoDB Atlas Vector Search](/docs/integrations/providers/mongodb_atlas) to semantically cache prompts and responses."
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "ecdc2a0a",
+   "metadata": {},
+   "source": [
+    "### `MongoDBCache`\n",
+    "An abstraction to store a simple cache in MongoDB. This does not use Semantic Caching, nor does it require an index to be made on the collection before generation.\n",
+    "\n",
+    "To import this cache:\n",
+    "\n",
+    "```python\n",
+    "from langchain_mongodb.cache import MongoDBCache\n",
+    "```\n",
+    "\n",
+    "\n",
+    "To use this cache with your LLMs:\n",
+    "```python\n",
+    "from langchain_core.globals import set_llm_cache\n",
+    "\n",
+    "# use any embedding provider...\n",
+    "from tests.integration_tests.vectorstores.fake_embeddings import FakeEmbeddings\n",
+    "\n",
+    "mongodb_atlas_uri = \"<YOUR_CONNECTION_STRING>\"\n",
+    "COLLECTION_NAME=\"<YOUR_CACHE_COLLECTION_NAME>\"\n",
+    "DATABASE_NAME=\"<YOUR_DATABASE_NAME>\"\n",
+    "\n",
+    "set_llm_cache(MongoDBCache(\n",
+    "    connection_string=mongodb_atlas_uri,\n",
+    "    collection_name=COLLECTION_NAME,\n",
+    "    database_name=DATABASE_NAME,\n",
+    "))\n",
+    "```\n",
+    "\n",
+    "\n",
+    "### `MongoDBAtlasSemanticCache`\n",
+    "Semantic caching allows users to retrieve cached prompts based on semantic similarity between the user input and previously cached results. Under the hood it blends MongoDBAtlas as both a cache and a vectorstore.\n",
+    "The MongoDBAtlasSemanticCache inherits from `MongoDBAtlasVectorSearch` and needs an Atlas Vector Search Index defined to work. Please look at the [usage example](/docs/integrations/vectorstores/mongodb_atlas) on how to set up the index.\n",
+    "\n",
+    "To import this cache:\n",
+    "```python\n",
+    "from langchain_mongodb.cache import MongoDBAtlasSemanticCache\n",
+    "```\n",
+    "\n",
+    "To use this cache with your LLMs:\n",
+    "```python\n",
+    "from langchain_core.globals import set_llm_cache\n",
+    "\n",
+    "# use any embedding provider...\n",
+    "from tests.integration_tests.vectorstores.fake_embeddings import FakeEmbeddings\n",
+    "\n",
+    "mongodb_atlas_uri = \"<YOUR_CONNECTION_STRING>\"\n",
+    "COLLECTION_NAME=\"<YOUR_CACHE_COLLECTION_NAME>\"\n",
+    "DATABASE_NAME=\"<YOUR_DATABASE_NAME>\"\n",
+    "\n",
+    "set_llm_cache(MongoDBAtlasSemanticCache(\n",
+    "    embedding=FakeEmbeddings(),\n",
+    "    connection_string=mongodb_atlas_uri,\n",
+    "    collection_name=COLLECTION_NAME,\n",
+    "    database_name=DATABASE_NAME,\n",
+    "))\n",
+    "```\n",
+    "\n",
+    "To find more resources about using MongoDBSemanticCache visit [here](https://www.mongodb.com/blog/post/introducing-semantic-caching-dedicated-mongodb-lang-chain-package-gen-ai-apps)"
+   ]
+  },
  {
   "cell_type": "markdown",
   "id": "726fe754",
@@ -993,7 +1070,7 @@
   "metadata": {},
   "outputs": [
    {
-     "name": "stdin",
+     "name": "stdout",
     "output_type": "stream",
     "text": [
      "CASSANDRA_KEYSPACE =  demo_keyspace\n"
@@ -1029,7 +1106,7 @@
   "metadata": {},
   "outputs": [
    {
-     "name": "stdin",
+     "name": "stdout",
     "output_type": "stream",
     "text": [
      "ASTRA_DB_ID =  01234567-89ab-cdef-0123-456789abcdef\n",
@@ -2071,6 +2148,71 @@
    "# so it uses the cached result!\n",
    "llm(\"Tell me one joke\")"
   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "ae1f5e1c-085e-4998-9f2d-b5867d2c3d5b",
+   "metadata": {
+    "execution": {
+     "iopub.execute_input": "2024-05-31T17:18:43.345495Z",
+     "iopub.status.busy": "2024-05-31T17:18:43.345015Z",
+     "iopub.status.idle": "2024-05-31T17:18:43.351003Z",
+     "shell.execute_reply": "2024-05-31T17:18:43.350073Z",
+     "shell.execute_reply.started": "2024-05-31T17:18:43.345456Z"
+    }
+   },
+   "source": [
+    "## Cache classes: summary table"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "65072e45-10bc-40f1-979b-2617656bbbce",
+   "metadata": {
+    "execution": {
+     "iopub.execute_input": "2024-05-31T17:16:05.616430Z",
+     "iopub.status.busy": "2024-05-31T17:16:05.616221Z",
+     "iopub.status.idle": "2024-05-31T17:16:05.624164Z",
+     "shell.execute_reply": "2024-05-31T17:16:05.623673Z",
+     "shell.execute_reply.started": "2024-05-31T17:16:05.616418Z"
+    }
+   },
+   "source": [
+    "**Cache** classes are implemented by inheriting the [BaseCache](https://api.python.langchain.com/en/latest/caches/langchain_core.caches.BaseCache.html) class.\n",
+    "\n",
+    "This table lists all 20 derived classes with links to the API Reference.\n",
+    "\n",
+    "\n",
+    "| Namespace 🔻 | Class |\n",
+    "|------------|---------|\n",
+    "| langchain_astradb.cache | [AstraDBCache](https://api.python.langchain.com/en/latest/cache/langchain_astradb.cache.AstraDBCache.html) |\n",
+    "| langchain_astradb.cache | [AstraDBSemanticCache](https://api.python.langchain.com/en/latest/cache/langchain_astradb.cache.AstraDBSemanticCache.html) |\n",
+    "| langchain_community.cache | [AstraDBCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.AstraDBCache.html) |\n",
+    "| langchain_community.cache | [AstraDBSemanticCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.AstraDBSemanticCache.html) |\n",
+    "| langchain_community.cache | [AzureCosmosDBSemanticCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.AzureCosmosDBSemanticCache.html) |\n",
+    "| langchain_community.cache | [CassandraCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.CassandraCache.html) |\n",
+    "| langchain_community.cache | [CassandraSemanticCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.CassandraSemanticCache.html) |\n",
+    "| langchain_community.cache | [GPTCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.GPTCache.html) |\n",
+    "| langchain_community.cache | [InMemoryCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.InMemoryCache.html) |\n",
+    "| langchain_community.cache | [MomentoCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.MomentoCache.html) |\n",
+    "| langchain_community.cache | [OpenSearchSemanticCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.OpenSearchSemanticCache.html) |\n",
+    "| langchain_community.cache | [RedisSemanticCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.RedisSemanticCache.html) |\n",
+    "| langchain_community.cache | [SQLAlchemyCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.SQLAlchemyCache.html) |\n",
+    "| langchain_community.cache | [SQLAlchemyMd5Cache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.SQLAlchemyMd5Cache.html) |\n",
+    "| langchain_community.cache | [UpstashRedisCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.UpstashRedisCache.html) |\n",
+    "| langchain_core.caches | [InMemoryCache](https://api.python.langchain.com/en/latest/caches/langchain_core.caches.InMemoryCache.html) |\n",
+    "| langchain_elasticsearch.cache | [ElasticsearchCache](https://api.python.langchain.com/en/latest/cache/langchain_elasticsearch.cache.ElasticsearchCache.html) |\n",
+    "| langchain_mongodb.cache | [MongoDBAtlasSemanticCache](https://api.python.langchain.com/en/latest/cache/langchain_mongodb.cache.MongoDBAtlasSemanticCache.html) |\n",
+    "| langchain_mongodb.cache | [MongoDBCache](https://api.python.langchain.com/en/latest/cache/langchain_mongodb.cache.MongoDBCache.html) |\n"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "19067f14-c69a-4156-9504-af43a0713669",
+   "metadata": {},
+   "outputs": [],
+   "source": []
  }
 ],
 "metadata": {
@@ -2089,7 +2231,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
-   "version": "3.9.1"
+   "version": "3.10.12"
  }
 },
 "nbformat": 4,
--- a/docs/docs/integrations/llms/aleph_alpha.ipynb
+++ b/docs/docs/integrations/llms/aleph_alpha.ipynb
@@ -12,6 +12,17 @@
    "This example goes over how to use LangChain to interact with Aleph Alpha models"
   ]
  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "84483bd5",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "# Installing the langchain package needed to use the integration\n",
+    "%pip install -qU langchain-community"
+   ]
+  },
  {
   "cell_type": "code",
   "execution_count": null,
--- a/docs/docs/integrations/llms/alibabacloud_pai_eas_endpoint.ipynb
+++ b/docs/docs/integrations/llms/alibabacloud_pai_eas_endpoint.ipynb
@@ -9,6 +9,16 @@
    ">[Machine Learning Platform for AI of Alibaba Cloud](https://www.alibabacloud.com/help/en/pai) is a machine learning or deep learning engineering platform intended for enterprises and developers. It provides easy-to-use, cost-effective, high-performance, and easy-to-scale plug-ins that can be applied to various industry scenarios. With over 140 built-in optimization algorithms, `Machine Learning Platform for AI` provides whole-process AI engineering capabilities including data labeling (`PAI-iTAG`), model building (`PAI-Designer` and `PAI-DSW`), model training (`PAI-DLC`), compilation optimization, and inference deployment (`PAI-EAS`). `PAI-EAS` supports different types of hardware resources, including CPUs and GPUs, and features high throughput and low latency. It allows you to deploy large-scale complex models with a few clicks and perform elastic scale-ins and scale-outs in real time. It also provides a comprehensive O&M and monitoring system."
   ]
  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "##Installing the langchain packages needed to use the integration\n",
+    "%pip install -qU langchain-community"
+   ]
+  },
  {
   "cell_type": "code",
   "execution_count": 8,
--- a/docs/docs/integrations/llms/amazon_api_gateway.ipynb
+++ b/docs/docs/integrations/llms/amazon_api_gateway.ipynb
@@ -16,6 +16,16 @@
    ">`API Gateway` handles all the tasks involved in accepting and processing up to hundreds of thousands of concurrent API calls, including traffic management, CORS support, authorization >and access control, throttling, monitoring, and API version management. `API Gateway` has no minimum fees or startup costs. You pay for the API calls you receive and the amount of data >transferred out and, with the `API Gateway` tiered pricing model, you can reduce your cost as your API usage scales."
   ]
  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "##Installing the langchain packages needed to use the integration\n",
+    "%pip install -qU langchain-community"
+   ]
+  },
  {
   "cell_type": "markdown",
   "metadata": {},
--- a/docs/docs/integrations/llms/anthropic.ipynb
+++ b/docs/docs/integrations/llms/anthropic.ipynb
@@ -3,10 +3,15 @@
  {
   "cell_type": "raw",
   "id": "602a52a4",
-   "metadata": {},
+   "metadata": {
+    "vscode": {
+     "languageId": "raw"
+    }
+   },
   "source": [
    "---\n",
    "sidebar_label: Anthropic\n",
+    "sidebar_class_name: hidden\n",
    "---"
   ]
  },
@@ -17,9 +22,13 @@
   "source": [
    "# AnthropicLLM\n",
    "\n",
-    "This example goes over how to use LangChain to interact with `Anthropic` models.\n",
+    ":::caution\n",
+    "You are currently on a page documenting the use of Anthropic legacy Claude 2 models as [text completion models](/docs/concepts/#llms). The latest and most popular Anthropic models are [chat completion models](/docs/concepts/#chat-models).\n",
    "\n",
-    "NOTE: AnthropicLLM only supports legacy Claude 2 models. To use the newest Claude 3 models, please use [`ChatAnthropic`](/docs/integrations/chat/anthropic) instead.\n",
+    "You are probably looking for [this page instead](/docs/integrations/chat/anthropic/).\n",
+    ":::\n",
+    "\n",
+    "This example goes over how to use LangChain to interact with `Anthropic` models.\n",
    "\n",
    "## Installation"
   ]
--- a/docs/docs/integrations/llms/anyscale.ipynb
+++ b/docs/docs/integrations/llms/anyscale.ipynb
@@ -12,6 +12,17 @@
    "This example goes over how to use LangChain to interact with [Anyscale Endpoint](https://app.endpoints.anyscale.com/). "
   ]
  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "134bd228",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "##Installing the langchain packages needed to use the integration\n",
+    "%pip install -qU langchain-community"
+   ]
+  },
  {
   "cell_type": "code",
   "execution_count": null,
--- a/docs/docs/integrations/llms/aphrodite.ipynb
+++ b/docs/docs/integrations/llms/aphrodite.ipynb
@@ -18,6 +18,17 @@
    "To use, you should have the `aphrodite-engine` python package installed."
   ]
  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "4dba1074",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "##Installing the langchain packages needed to use the integration\n",
+    "%pip install -qU langchain-community"
+   ]
+  },
  {
   "cell_type": "code",
   "execution_count": null,
--- a/docs/docs/integrations/llms/arcee.ipynb
+++ b/docs/docs/integrations/llms/arcee.ipynb
@@ -8,6 +8,16 @@
    "This notebook demonstrates how to use the `Arcee` class for generating text using Arcee's Domain Adapted Language Models (DALMs)."
   ]
  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "##Installing the langchain packages needed to use the integration\n",
+    "%pip install -qU langchain-community"
+   ]
+  },
  {
   "cell_type": "markdown",
   "metadata": {},
--- a/docs/docs/integrations/llms/azure_ml.ipynb
+++ b/docs/docs/integrations/llms/azure_ml.ipynb
@@ -11,6 +11,16 @@
    "This notebook goes over how to use an LLM hosted on an `Azure ML Online Endpoint`."
   ]
  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "##Installing the langchain packages needed to use the integration\n",
+    "%pip install -qU langchain-community"
+   ]
+  },
  {
   "cell_type": "code",
   "execution_count": null,
--- a/docs/docs/integrations/llms/azure_openai.ipynb
+++ b/docs/docs/integrations/llms/azure_openai.ipynb
@@ -7,7 +7,13 @@
   "source": [
    "# Azure OpenAI\n",
    "\n",
-    "This notebook goes over how to use Langchain with [Azure OpenAI](https://aka.ms/azure-openai).\n",
+    ":::caution\n",
+    "You are currently on a page documenting the use of Azure OpenAI [text completion models](/docs/concepts/#llms). The latest and most popular Azure OpenAI models are [chat completion models](/docs/concepts/#chat-models).\n",
+    "\n",
+    "Unless you are specifically using `gpt-3.5-turbo-instruct`, you are probably looking for [this page instead](/docs/integrations/chat/azure_chat_openai/).\n",
+    ":::\n",
+    "\n",
+    "This page goes over how to use LangChain with [Azure OpenAI](https://aka.ms/azure-openai).\n",
    "\n",
    "The Azure OpenAI API is compatible with OpenAI's API.  The `openai` Python package makes it easy to use both OpenAI and Azure OpenAI.  You can call Azure OpenAI the same way you call OpenAI with the exceptions noted below.\n",
    "\n",
--- a/docs/docs/integrations/llms/baichuan.ipynb
+++ b/docs/docs/integrations/llms/baichuan.ipynb
@@ -8,6 +8,16 @@
    "Baichuan Inc. (https://www.baichuan-ai.com/) is a Chinese startup in the era of AGI, dedicated to addressing fundamental human needs: Efficiency, Health, and Happiness."
   ]
  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "##Installing the langchain packages needed to use the integration\n",
+    "%pip install -qU langchain-community"
+   ]
+  },
  {
   "cell_type": "markdown",
   "metadata": {},
--- a/docs/docs/integrations/llms/baidu_qianfan_endpoint.ipynb
+++ b/docs/docs/integrations/llms/baidu_qianfan_endpoint.ipynb
@@ -45,6 +45,16 @@
    "- AquilaChat-7B"
   ]
  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "##Installing the langchain packages needed to use the integration\n",
+    "%pip install -qU langchain-community"
+   ]
+  },
  {
   "cell_type": "code",
   "execution_count": 2,
--- a/docs/docs/integrations/llms/banana.ipynb
+++ b/docs/docs/integrations/llms/banana.ipynb
@@ -12,6 +12,16 @@
    "This example goes over how to use LangChain to interact with Banana models"
   ]
  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "##Installing the langchain packages needed to use the integration\n",
+    "%pip install -qU  langchain-community"
+   ]
+  },
  {
   "cell_type": "code",
   "execution_count": null,
--- a/docs/docs/integrations/llms/baseten.ipynb
+++ b/docs/docs/integrations/llms/baseten.ipynb
@@ -45,6 +45,16 @@
    "In this example, we'll work with Mistral 7B. [Deploy Mistral 7B here](https://app.baseten.co/explore/mistral_7b_instruct) and follow along with the deployed model's ID, found in the model dashboard."
   ]
  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "##Installing the langchain packages needed to use the integration\n",
+    "%pip install -qU langchain-community"
+   ]
+  },
  {
   "cell_type": "code",
   "execution_count": null,
--- a/docs/docs/integrations/llms/bedrock.ipynb
+++ b/docs/docs/integrations/llms/bedrock.ipynb
@@ -11,6 +11,12 @@
   "cell_type": "markdown",
   "metadata": {},
   "source": [
+    ":::caution\n",
+    "You are currently on a page documenting the use of Amazon Bedrock models as [text completion models](/docs/concepts/#llms). Many popular models available on Bedrock are [chat completion models](/docs/concepts/#chat-models).\n",
+    "\n",
+    "You may be looking for [this page instead](/docs/integrations/chat/bedrock/).\n",
+    ":::\n",
+    "\n",
    ">[Amazon Bedrock](https://aws.amazon.com/bedrock/) is a fully managed service that offers a choice of \n",
    "> high-performing foundation models (FMs) from leading AI companies like `AI21 Labs`, `Anthropic`, `Cohere`, \n",
    "> `Meta`, `Stability AI`, and `Amazon` via a single API, along with a broad set of capabilities you need to \n",
--- a/docs/docs/integrations/llms/cohere.ipynb
+++ b/docs/docs/integrations/llms/cohere.ipynb
@@ -7,6 +7,12 @@
   "source": [
    "# Cohere\n",
    "\n",
+    ":::caution\n",
+    "You are currently on a page documenting the use of Cohere models as [text completion models](/docs/concepts/#llms). Many popular Cohere models are [chat completion models](/docs/concepts/#chat-models).\n",
+    "\n",
+    "You may be looking for [this page instead](/docs/integrations/chat/cohere/).\n",
+    ":::\n",
+    "\n",
    ">[Cohere](https://cohere.ai/about) is a Canadian startup that provides natural language processing models that help companies improve human-machine interactions.\n",
    "\n",
    "Head to the [API reference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.cohere.Cohere.html) for detailed documentation of all attributes and methods."
@@ -193,7 +199,7 @@
   "id": "39198f7d-6fc8-4662-954a-37ad38c4bec4",
   "metadata": {},
   "source": [
-    "You can also easily combine with a prompt template for easy structuring of user input. We can do this using [LCEL](/docs/concepts#langchain-expression-language)"
+    "You can also easily combine with a prompt template for easy structuring of user input. We can do this using [LCEL](/docs/concepts#langchain-expression-language-lcel)"
   ]
  },
  {
--- a/docs/docs/integrations/llms/fireworks.ipynb
+++ b/docs/docs/integrations/llms/fireworks.ipynb
@@ -7,6 +7,12 @@
   "source": [
    "# Fireworks\n",
    "\n",
+    ":::caution\n",
+    "You are currently on a page documenting the use of Fireworks models as [text completion models](/docs/concepts/#llms). Many popular Fireworks models are [chat completion models](/docs/concepts/#chat-models).\n",
+    "\n",
+    "You may be looking for [this page instead](/docs/integrations/chat/fireworks/).\n",
+    ":::\n",
+    "\n",
    ">[Fireworks](https://app.fireworks.ai/) accelerates product development on generative AI by creating an innovative AI experiment and production platform. \n",
    "\n",
    "This example goes over how to use LangChain to interact with `Fireworks` models."
--- a/docs/docs/integrations/llms/google_ai.ipynb
+++ b/docs/docs/integrations/llms/google_ai.ipynb
@@ -25,6 +25,12 @@
   "id": "bead5ede-d9cc-44b9-b062-99c90a10cf40",
   "metadata": {},
   "source": [
+    ":::caution\n",
+    "You are currently on a page documenting the use of Google models as [text completion models](/docs/concepts/#llms). Many popular Google models are [chat completion models](/docs/concepts/#chat-models).\n",
+    "\n",
+    "You may be looking for [this page instead](/docs/integrations/chat/google_generative_ai/).\n",
+    ":::\n",
+    "\n",
    "A guide on using [Google Generative AI](https://developers.generativeai.google/) models with Langchain. Note: It's separate from Google Cloud Vertex AI [integration](/docs/integrations/llms/google_vertex_ai_palm)."
   ]
  },
--- a/docs/docs/integrations/llms/google_vertex_ai_palm.ipynb
+++ b/docs/docs/integrations/llms/google_vertex_ai_palm.ipynb
@@ -15,6 +15,12 @@
   "source": [
    "# Google Cloud Vertex AI\n",
    "\n",
+    ":::caution\n",
+    "You are currently on a page documenting the use of Google Vertex [text completion models](/docs/concepts/#llms). Many Google models are [chat completion models](/docs/concepts/#chat-models).\n",
+    "\n",
+    "You may be looking for [this page instead](/docs/integrations/chat/google_vertex_ai_palm/).\n",
+    ":::\n",
+    "\n",
    "**Note:** This is separate from the `Google Generative AI` integration, it exposes [Vertex AI Generative API](https://cloud.google.com/vertex-ai/docs/generative-ai/learn/overview) on `Google Cloud`.\n",
    "\n",
    "VertexAI exposes all foundational models available in google cloud:\n",
@@ -328,7 +334,7 @@
   "cell_type": "markdown",
   "metadata": {},
   "source": [
-    "You can also easily combine with a prompt template for easy structuring of user input. We can do this using [LCEL](/docs/concepts#langchain-expression-language)"
+    "You can also easily combine with a prompt template for easy structuring of user input. We can do this using [LCEL](/docs/concepts#langchain-expression-language-lcel)"
   ]
  },
  {
--- a/docs/docs/integrations/llms/huggingface_pipelines.ipynb
+++ b/docs/docs/integrations/llms/huggingface_pipelines.ipynb
@@ -121,6 +121,28 @@
    "print(chain.invoke({\"question\": question}))"
   ]
  },
+  {
+   "cell_type": "markdown",
+   "id": "b4a31db5",
+   "metadata": {},
+   "source": [
+    "To get response without prompt, you can bind `skip_prompt=True` with LLM."
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "5e4aaad2",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "chain = prompt | hf.bind(skip_prompt=True)\n",
+    "\n",
+    "question = \"What is electroencephalography?\"\n",
+    "\n",
+    "print(chain.invoke({\"question\": question}))"
+   ]
+  },
  {
   "cell_type": "markdown",
   "id": "dbbc3a37",
--- a/docs/docs/integrations/llms/ollama.ipynb
+++ b/docs/docs/integrations/llms/ollama.ipynb
@@ -6,6 +6,12 @@
   "source": [
    "# Ollama\n",
    "\n",
+    ":::caution\n",
+    "You are currently on a page documenting the use of Ollama models as [text completion models](/docs/concepts/#llms). Many popular Ollama models are [chat completion models](/docs/concepts/#chat-models).\n",
+    "\n",
+    "You may be looking for [this page instead](/docs/integrations/chat/ollama/).\n",
+    ":::\n",
+    "\n",
    "[Ollama](https://ollama.ai/) allows you to run open-source large language models, such as Llama 2, locally.\n",
    "\n",
    "Ollama bundles model weights, configuration, and data into a single package, defined by a Modelfile. \n",
--- a/docs/docs/integrations/llms/openai.ipynb
+++ b/docs/docs/integrations/llms/openai.ipynb
@@ -7,6 +7,12 @@
   "source": [
    "# OpenAI\n",
    "\n",
+    ":::caution\n",
+    "You are currently on a page documenting the use of OpenAI [text completion models](/docs/concepts/#llms). The latest and most popular OpenAI models are [chat completion models](/docs/concepts/#chat-models).\n",
+    "\n",
+    "Unless you are specifically using `gpt-3.5-turbo-instruct`, you are probably looking for [this page instead](/docs/integrations/chat/openai/).\n",
+    ":::\n",
+    "\n",
    "[OpenAI](https://platform.openai.com/docs/introduction) offers a spectrum of models with different levels of power suitable for different tasks.\n",
    "\n",
    "This example goes over how to use LangChain to interact with `OpenAI` [models](https://platform.openai.com/docs/models)"
--- a/docs/docs/integrations/llms/openvino.ipynb
+++ b/docs/docs/integrations/llms/openvino.ipynb
@@ -31,7 +31,7 @@
   },
   "outputs": [],
   "source": [
-    "%pip install --upgrade-strategy eager \"optimum[openvino,nncf]\" --quiet"
+    "%pip install --upgrade-strategy eager \"optimum[openvino,nncf]\" langchain-huggingface --quiet"
   ]
  },
  {
@@ -130,6 +130,28 @@
    "print(chain.invoke({\"question\": question}))"
   ]
  },
+  {
+   "cell_type": "markdown",
+   "id": "446a01e0",
+   "metadata": {},
+   "source": [
+    "To get response without prompt, you can bind `skip_prompt=True` with LLM."
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "e3baeab2",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "chain = prompt | ov_llm.bind(skip_prompt=True)\n",
+    "\n",
+    "question = \"What is electroencephalography?\"\n",
+    "\n",
+    "print(chain.invoke({\"question\": question}))"
+   ]
+  },
  {
   "cell_type": "markdown",
   "id": "12524837-e9ab-455a-86be-66b95f4f893a",
@@ -243,7 +265,8 @@
    "    skip_prompt=True,\n",
    "    skip_special_tokens=True,\n",
    ")\n",
-    "ov_llm.pipeline._forward_params = {\"streamer\": streamer, \"max_new_tokens\": 100}\n",
+    "pipeline_kwargs = {\"pipeline_kwargs\": {\"streamer\": streamer, \"max_new_tokens\": 100}}\n",
+    "chain = prompt | ov_llm.bind(**pipeline_kwargs)\n",
    "\n",
    "t1 = Thread(target=chain.invoke, args=({\"question\": question},))\n",
    "t1.start()\n",
--- a/docs/docs/integrations/llms/together.ipynb
+++ b/docs/docs/integrations/llms/together.ipynb
@@ -7,6 +7,12 @@
   "source": [
    "# Together AI\n",
    "\n",
+    ":::caution\n",
+    "You are currently on a page documenting the use of Together AI models as [text completion models](/docs/concepts/#llms). Many popular Together AI models are [chat completion models](/docs/concepts/#chat-models).\n",
+    "\n",
+    "You may be looking for [this page instead](/docs/integrations/chat/together/).\n",
+    ":::\n",
+    "\n",
    "[Together AI](https://www.together.ai/) offers an API to query [50+ leading open-source models](https://docs.together.ai/docs/inference-models) in a couple lines of code.\n",
    "\n",
    "This example goes over how to use LangChain to interact with Together AI models."
--- a/docs/docs/integrations/platforms/anthropic.mdx
+++ b/docs/docs/integrations/platforms/anthropic.mdx
@@ -14,6 +14,18 @@ pip install -U langchain-anthropic
 You need to set the `ANTHROPIC_API_KEY` environment variable.
 You can get an Anthropic API key [here](https://console.anthropic.com/settings/keys)

+## Chat Models
+
+### ChatAnthropic
+
+See a [usage example](/docs/integrations/chat/anthropic).
+
+```python
+from langchain_anthropic import ChatAnthropic
+
+model = ChatAnthropic(model='claude-3-opus-20240229')
+```
+
 ## LLMs

 ### [Legacy] AnthropicLLM
@@ -28,17 +40,3 @@ from langchain_anthropic import AnthropicLLM

 model = AnthropicLLM(model='claude-2.1')
 ```
-
-## Chat Models
-
-### ChatAnthropic
-
-See a [usage example](/docs/integrations/chat/anthropic).
-
-```python
-from langchain_anthropic import ChatAnthropic
-
-model = ChatAnthropic(model='claude-3-opus-20240229')
-```
-
-
--- a/docs/docs/integrations/platforms/google.mdx
+++ b/docs/docs/integrations/platforms/google.mdx
@@ -2,45 +2,10 @@

 All functionality related to [Google Cloud Platform](https://cloud.google.com/) and other `Google` products.

-## LLMs
+## Chat models

 We recommend individual developers to start with Gemini API (`langchain-google-genai`) and move to Vertex AI (`langchain-google-vertexai`) when they need access to commercial support and higher rate limits. If you’re already Cloud-friendly or Cloud-native, then you can get started in Vertex AI straight away.
-Please, find more information [here](https://ai.google.dev/gemini-api/docs/migrate-to-cloud).
-
-### Google Generative AI
-
-Access GoogleAI `Gemini` models such as `gemini-pro` and `gemini-pro-vision` through the `GoogleGenerativeAI` class.
-
-Install python package.
-
-```bash
-pip install langchain-google-genai
-```
-
-See a [usage example](/docs/integrations/llms/google_ai).
-
-```python
-from langchain_google_genai import GoogleGenerativeAI
-```
-
-### Vertex AI Model Garden
-
-Access `PaLM` and hundreds of OSS models via `Vertex AI Model Garden` service.
-
-We need to install `langchain-google-vertexai` python package.
-
-```bash
-pip install langchain-google-vertexai
-```
-
-See a [usage example](/docs/integrations/llms/google_vertex_ai_palm#vertex-model-garden).
-
-```python
-from langchain_google_vertexai import VertexAIModelGarden
-```
-
-
-## Chat models
+Please see [here](https://ai.google.dev/gemini-api/docs/migrate-to-cloud) for more information.

 ### Google Generative AI

@@ -107,6 +72,40 @@ See a [usage example](/docs/integrations/chat/google_vertex_ai_palm).
 from langchain_google_vertexai import ChatVertexAI
 ```

+## LLMs
+
+### Google Generative AI
+
+Access GoogleAI `Gemini` models such as `gemini-pro` and `gemini-pro-vision` through the `GoogleGenerativeAI` class.
+
+Install python package.
+
+```bash
+pip install langchain-google-genai
+```
+
+See a [usage example](/docs/integrations/llms/google_ai).
+
+```python
+from langchain_google_genai import GoogleGenerativeAI
+```
+
+### Vertex AI Model Garden
+
+Access `PaLM` and hundreds of OSS models via `Vertex AI Model Garden` service.
+
+We need to install `langchain-google-vertexai` python package.
+
+```bash
+pip install langchain-google-vertexai
+```
+
+See a [usage example](/docs/integrations/llms/google_vertex_ai_palm#vertex-model-garden).
+
+```python
+from langchain_google_vertexai import VertexAIModelGarden
+```
+
 ## Embedding models

 ### Google Generative AI Embeddings
--- a/docs/docs/integrations/platforms/index.mdx
+++ b/docs/docs/integrations/platforms/index.mdx
@@ -24,6 +24,7 @@ These providers have standalone `langchain-{provider}` packages for improved ver
 - [Anthropic](/docs/integrations/platforms/anthropic)
 - [Astra DB](/docs/integrations/providers/astradb)
 - [Cohere](/docs/integrations/providers/cohere)
+- [Couchbase](/docs/integrations/providers/couchbase)
 - [Elasticsearch](/docs/integrations/providers/elasticsearch)
 - [Exa Search](/docs/integrations/providers/exa_search)
 - [Fireworks](/docs/integrations/providers/fireworks)
--- a/docs/docs/integrations/platforms/microsoft.mdx
+++ b/docs/docs/integrations/platforms/microsoft.mdx
@@ -6,24 +6,6 @@ keywords: [azure]

 All functionality related to `Microsoft Azure` and other `Microsoft` products.

-## LLMs
-
-### Azure ML
-
-See a [usage example](/docs/integrations/llms/azure_ml).
-
-```python
-from langchain_community.llms.azureml_endpoint import AzureMLOnlineEndpoint
-```
-
-### Azure OpenAI
-
-See a [usage example](/docs/integrations/llms/azure_openai).
-
-```python
-from langchain_openai import AzureOpenAI
-```
-
 ## Chat Models
 ### Azure OpenAI

@@ -51,6 +33,24 @@ See a [usage example](/docs/integrations/chat/azure_chat_openai)
 from langchain_openai import AzureChatOpenAI
 ```

+## LLMs
+
+### Azure ML
+
+See a [usage example](/docs/integrations/llms/azure_ml).
+
+```python
+from langchain_community.llms.azureml_endpoint import AzureMLOnlineEndpoint
+```
+
+### Azure OpenAI
+
+See a [usage example](/docs/integrations/llms/azure_openai).
+
+```python
+from langchain_openai import AzureOpenAI
+```
+
 ## Embedding Models
 ### Azure OpenAI

@@ -225,7 +225,7 @@ from langchain_community.document_loaders.onenote import OneNoteLoader

 ## Vector stores

-### Azure Cosmos DB
+### Azure Cosmos DB MongoDB vCore

 >[Azure Cosmos DB for MongoDB vCore](https://learn.microsoft.com/en-us/azure/cosmos-db/mongodb/vcore/) makes it easy to create a database with full native MongoDB support.
 > You can apply your MongoDB experience and continue to use your favorite MongoDB drivers, SDKs, and tools by pointing your application to the API for MongoDB vCore account's connection string.
@@ -255,6 +255,38 @@ See a [usage example](/docs/integrations/vectorstores/azure_cosmos_db).
 from langchain_community.vectorstores import AzureCosmosDBVectorSearch
 ```

+### Azure Cosmos DB NoSQL
+
+>[Azure Cosmos DB for NoSQL](https://learn.microsoft.com/en-us/azure/cosmos-db/nosql/vector-search) now offers vector indexing and search in preview.
+This feature is designed to handle high-dimensional vectors, enabling efficient and accurate vector search at any scale. You can now store vectors
+directly in the documents alongside your data. This means that each document in your database can contain not only traditional schema-free data,
+but also high-dimensional vectors as other properties of the documents. This colocation of data and vectors allows for efficient indexing and searching,
+as the vectors are stored in the same logical unit as the data they represent. This simplifies data management, AI application architectures, and the
+efficiency of vector-based operations.
+
+#### Installation and Setup
+
+See [detail configuration instructions](/docs/integrations/vectorstores/azure_cosmos_db_no_sql).
+
+We need to install `azure-cosmos` python package.
+
+```bash
+pip install azure-cosmos
+```
+
+#### Deploy Azure Cosmos DB on Microsoft Azure
+
+Azure Cosmos DB offers a solution for modern apps and intelligent workloads by being very responsive with dynamic and elastic autoscale. It is available
+in every Azure region and can automatically replicate data closer to users. It has SLA guaranteed low-latency and high availability.
+
+[Sign Up](https://learn.microsoft.com/en-us/azure/cosmos-db/nosql/quickstart-python?pivots=devcontainer-codespace) for free to get started today.
+
+See a [usage example](/docs/integrations/vectorstores/azure_cosmos_db_no_sql).
+
+```python
+from langchain_community.vectorstores import AzureCosmosDBNoSQLVectorSearch
+```
+
 ## Retrievers
 ### Azure AI Search

--- a/docs/docs/integrations/platforms/openai.mdx
+++ b/docs/docs/integrations/platforms/openai.mdx
@@ -25,6 +25,19 @@ pip install langchain-openai

 Get an OpenAI api key and set it as an environment variable (`OPENAI_API_KEY`)

+## Chat model
+
+See a [usage example](/docs/integrations/chat/openai).
+
+```python
+from langchain_openai import ChatOpenAI
+```
+
+If you are using a model hosted on `Azure`, you should use different wrapper for that:
+```python
+from langchain_openai import AzureChatOpenAI
+```
+For a more detailed walkthrough of the `Azure` wrapper, see [here](/docs/integrations/chat/azure_chat_openai).

 ## LLM

@@ -38,21 +51,7 @@ If you are using a model hosted on `Azure`, you should use different wrapper for
 ```python
 from langchain_openai import AzureOpenAI
 ```
-For a more detailed walkthrough of the `Azure` wrapper, see [here](/docs/integrations/llms/azure_openai)
-
-## Chat model
-
-See a [usage example](/docs/integrations/chat/openai).
-
-```python
-from langchain_openai import ChatOpenAI
-```
-
-If you are using a model hosted on `Azure`, you should use different wrapper for that:
-```python
-from langchain_openai import AzureChatOpenAI
-```
-For a more detailed walkthrough of the `Azure` wrapper, see [here](/docs/integrations/chat/azure_chat_openai)
+For a more detailed walkthrough of the `Azure` wrapper, see [here](/docs/integrations/llms/azure_openai).

 ## Embedding Model

--- a/docs/docs/integrations/providers/couchbase.mdx
+++ b/docs/docs/integrations/providers/couchbase.mdx
@@ -6,12 +6,19 @@

 ## Installation and Setup

-We have to install the `couchbase`package.
+We have to install the `langchain-couchbase` package.

 ```bash
-pip install couchbase
+pip install langchain-couchbase
 ```

+## Vector Store
+
+See a [usage example](/docs/integrations/vectorstores/couchbase).
+
+```python
+from langchain_couchbase import CouchbaseVectorStore
+```

 ## Document loader

--- a/docs/docs/integrations/providers/vectara/index.mdx
+++ b/docs/docs/integrations/providers/vectara/index.mdx
@@ -1,28 +1,38 @@
 # Vectara

->[Vectara](https://vectara.com/) is the trusted GenAI platform for developers. It provides a simple API to build GenAI applications
-> for semantic search or RAG (Retreieval augmented generation).
+>[Vectara](https://vectara.com/) provides a Trusted Generative AI platform, allowing organizations to rapidly create a ChatGPT-like experience (an AI assistant) 
+> which is grounded in the data, documents, and knowledge that they have (technically, it is Retrieval-Augmented-Generation-as-a-service).

 **Vectara Overview:**
- `Vectara` is developer-first API platform for building trusted GenAI applications. 
- To use Vectara - first [sign up](https://vectara.com/integrations/langchain) and create an account. Then create a corpus and an API key for indexing and searching.
- You can use Vectara's [indexing API](https://docs.vectara.com/docs/indexing-apis/indexing) to add documents into Vectara's index
- You can use Vectara's [Search API](https://docs.vectara.com/docs/search-apis/search) to query Vectara's index (which also supports Hybrid search implicitly).
+`Vectara` is RAG-as-a-service, providing all the components of RAG behind an easy-to-use API, including:
+1. A way to extract text from files (PDF, PPT, DOCX, etc)
+2. ML-based chunking that provides state of the art performance.
+3. The [Boomerang](https://vectara.com/how-boomerang-takes-retrieval-augmented-generation-to-the-next-level-via-grounded-generation/) embeddings model.
+4. Its own internal vector database where text chunks and embedding vectors are stored.
+5. A query service that automatically encodes the query into embedding, and retrieves the most relevant text segments
+(including support for [Hybrid Search](https://docs.vectara.com/docs/api-reference/search-apis/lexical-matching) and 
+[MMR](https://vectara.com/get-diverse-results-and-comprehensive-summaries-with-vectaras-mmr-reranker/))
+7. An LLM to for creating a [generative summary](https://docs.vectara.com/docs/learn/grounded-generation/grounded-generation-overview), based on the retrieved documents (context), including citations.
+
+For more information:
+- [Documentation](https://docs.vectara.com/docs/)
+- [API Playground](https://docs.vectara.com/docs/rest-api/)
+- [Quickstart](https://docs.vectara.com/docs/quickstart)

 ## Installation and Setup

 To use `Vectara` with LangChain no special installation steps are required. 
-To get started, [sign up](https://vectara.com/integrations/langchain) and follow our [quickstart](https://docs.vectara.com/docs/quickstart) guide to create a corpus and an API key. 
-Once you have these, you can provide them as arguments to the Vectara vectorstore, or you can set them as environment variables.
+To get started, [sign up](https://vectara.com/integrations/langchain) for a free Vectara account (if you don't already have one), 
+and follow the [quickstart](https://docs.vectara.com/docs/quickstart) guide to create a corpus and an API key. 
+Once you have these, you can provide them as arguments to the Vectara `vectorstore`, or you can set them as environment variables.

 - export `VECTARA_CUSTOMER_ID`="your_customer_id"
 - export `VECTARA_CORPUS_ID`="your_corpus_id"
 - export `VECTARA_API_KEY`="your-vectara-api-key"

-
 ## Vectara as a Vector Store

-There exists a wrapper around the Vectara platform, allowing you to use it as a vectorstore, whether for semantic search or example selection.
+There exists a wrapper around the Vectara platform, allowing you to use it as a `vectorstore` in LangChain:

 To import this vectorstore:
 ```python
@@ -37,7 +47,10 @@ vectara = Vectara(
    vectara_api_key=api_key
 )
 ```
-The customer_id, corpus_id and api_key are optional, and if they are not supplied will be read from the environment variables `VECTARA_CUSTOMER_ID`, `VECTARA_CORPUS_ID` and `VECTARA_API_KEY`, respectively.
+The `customer_id`, `corpus_id` and `api_key` are optional, and if they are not supplied will be read from 
+the environment variables `VECTARA_CUSTOMER_ID`, `VECTARA_CORPUS_ID` and `VECTARA_API_KEY`, respectively.
+
+### Adding Texts or Files

 After you have the vectorstore, you can `add_texts` or `add_documents` as per the standard `VectorStore` interface, for example:

@@ -45,8 +58,8 @@ After you have the vectorstore, you can `add_texts` or `add_documents` as per th
 vectara.add_texts(["to be or not to be", "that is the question"])
 ```

-
-Since Vectara supports file-upload, we also added the ability to upload files (PDF, TXT, HTML, PPT, DOC, etc) directly as file. When using this method, the file is uploaded directly to the Vectara backend, processed and chunked optimally there, so you don't have to use the LangChain document loader or chunking mechanism.
+Since Vectara supports file-upload in the platform, we also added the ability to upload files (PDF, TXT, HTML, PPT, DOC, etc) directly. 
+When using this method, each file is uploaded directly to the Vectara backend, processed and chunked optimally there, so you don't have to use the LangChain document loader or chunking mechanism.

 As an example:

@@ -54,9 +67,13 @@ As an example:
 vectara.add_files(["path/to/file1.pdf", "path/to/file2.pdf",...])
 ```

-To query the vectorstore, you can use the `similarity_search` method (or `similarity_search_with_score`), which takes a query string and returns a list of results:
+Of course you do not have to add any data, and instead just connect to an existing Vectara corpus where data may already be indexed.
+
+### Querying the VectorStore
+
+To query the Vectara vectorstore, you can use the `similarity_search` method (or `similarity_search_with_score`), which takes a query string and returns a list of results:
 ```python
-results = vectara.similarity_score("what is LangChain?")
+results = vectara.similarity_search_with_score("what is LangChain?")
 ```
 The results are returned as a list of relevant documents, and a relevance score of each document.

@@ -65,28 +82,101 @@ In this case, we used the default retrieval parameters, but you can also specify
 - `lambda_val`: the [lexical matching](https://docs.vectara.com/docs/api-reference/search-apis/lexical-matching) factor for hybrid search (defaults to 0.025)
 - `filter`: a [filter](https://docs.vectara.com/docs/common-use-cases/filtering-by-metadata/filter-overview) to apply to the results (default None)
 - `n_sentence_context`: number of sentences to include before/after the actual matching segment when returning results. This defaults to 2.
- `mmr_config`: can be used to specify MMR mode in the query.
-   - `is_enabled`: True or False
-   - `mmr_k`: number of results to use for MMR reranking
-   - `diversity_bias`: 0 = no diversity, 1 = full diversity. This is the lambda parameter in the MMR formula and is in the range 0...1
+- `rerank_config`: can be used to specify reranker for thr results
+   - `reranker`: mmr, rerank_multilingual_v1 or none. Note that "rerank_multilingual_v1" is a Scale only feature
+   - `rerank_k`: number of results to use for reranking
+   - `mmr_diversity_bias`: 0 = no diversity, 1 = full diversity. This is the lambda parameter in the MMR formula and is in the range 0...1
+
+To get results without the relevance score, you can simply use the 'similarity_search' method:
+```python   
+results = vectara.similarity_search("what is LangChain?")
+```

 ## Vectara for Retrieval Augmented Generation (RAG)

-Vectara provides a full RAG pipeline, including generative summarization. 
-To use this pipeline, you can specify the `summary_config` argument in `similarity_search` or `similarity_search_with_score` as follows:
+Vectara provides a full RAG pipeline, including generative summarization. To use it as a complete RAG solution, you can use the `as_rag` method.
+There are a few additional parameters that can be specified in the `VectaraQueryConfig` object to control retrieval and summarization:
+* k: number of results to return
+* lambda_val: the lexical matching factor for hybrid search
+* summary_config (optional): can be used to request an LLM summary in RAG
+   - is_enabled: True or False
+   - max_results: number of results to use for summary generation
+   - response_lang: language of the response summary, in ISO 639-2 format (e.g. 'en', 'fr', 'de', etc)
+* rerank_config (optional): can be used to specify Vectara Reranker of the results
+   - reranker: mmr, rerank_multilingual_v1 or none
+   - rerank_k: number of results to use for reranking
+   - mmr_diversity_bias: 0 = no diversity, 1 = full diversity. 
+     This is the lambda parameter in the MMR formula and is in the range 0...1

- `summary_config`: can be used to request an LLM summary in RAG
-   - `is_enabled`: True or False
-   - `max_results`: number of results to use for summary generation
-   - `response_lang`: language of the response summary, in ISO 639-2 format (e.g. 'en', 'fr', 'de', etc)
+For example:
+
+```python
+summary_config = SummaryConfig(is_enabled=True, max_results=7, response_lang='eng')
+rerank_config = RerankConfig(reranker="mmr", rerank_k=50, mmr_diversity_bias=0.2)
+config = VectaraQueryConfig(k=10, lambda_val=0.005, rerank_config=rerank_config, summary_config=summary_config)
+```
+Then you can use the `as_rag` method to create a RAG pipeline:
+
+```python
+query_str = "what did Biden say?"
+
+rag = vectara.as_rag(config)
+rag.invoke(query_str)['answer']
+```
+
+The `as_rag` method returns a `VectaraRAG` object, which behaves just like any LangChain Runnable, including the `invoke` or `stream` methods.
+
+## Vectara Chat
+
+The RAG functionality can be used to create a chatbot. For example, you can create a simple chatbot that responds to user input:
+
+```python
+summary_config = SummaryConfig(is_enabled=True, max_results=7, response_lang='eng')
+rerank_config = RerankConfig(reranker="mmr", rerank_k=50, mmr_diversity_bias=0.2)
+config = VectaraQueryConfig(k=10, lambda_val=0.005, rerank_config=rerank_config, summary_config=summary_config)
+
+query_str = "what did Biden say?"
+bot = vectara.as_chat(config)
+bot.invoke(query_str)['answer']
+```
+
+The main difference is the following: with `as_chat` Vectara internally tracks the chat history and conditions each response on the full chat history.
+There is no need to keep that history locally to LangChain, as Vectara will manage it internally.
+
+## Vectara as a LangChain retriever only
+
+If you want to use Vectara as a retriever only, you can use the `as_retriever` method, which returns a `VectaraRetriever` object.
+```python
+retriever = vectara.as_retriever(config=config)
+retriever.invoke(query_str)
+```
+
+Like with as_rag, you provide a `VectaraQueryConfig` object to control the retrieval parameters.
+In most cases you would not enable the summary_config, but it is left as an option for backwards compatibility. 
+If no summary is requested, the response will be a list of relevant documents, each with a relevance score.
+If a summary is requested, the response will be a list of relevant documents as before, plus an additional document that includes the generative summary.
+
+## Hallucination Detection score
+
+Vectara created [HHEM](https://huggingface.co/vectara/hallucination_evaluation_model) - an open source model that can be used to evaluate RAG responses for factual consistency. 
+As part of the Vectara RAG, the "Factual Consistency Score" (or FCS), which is an improved version of the open source HHEM is made available via the API. 
+This is automatically included in the output of the RAG pipeline
+
+```python
+summary_config = SummaryConfig(is_enabled=True, max_results=7, response_lang='eng')
+rerank_config = RerankConfig(reranker="mmr", rerank_k=50, mmr_diversity_bias=0.2)
+config = VectaraQueryConfig(k=10, lambda_val=0.005, rerank_config=rerank_config, summary_config=summary_config)
+
+rag = vectara.as_rag(config)
+resp = rag.invoke(query_str)
+print(resp['answer'])
+print(f"Vectara FCS = {resp['fcs']}")
+```

 ## Example Notebooks

-For a more detailed examples of using Vectara, see the following examples:
-* [this notebook](/docs/integrations/vectorstores/vectara) shows how to use Vectara as a vectorstore for semantic search
-* [this notebook](/docs/integrations/providers/vectara/vectara_chat) shows how to build a chatbot with Langchain and Vectara
-* [this notebook](/docs/integrations/providers/vectara/vectara_summary) shows how to use the full Vectara RAG pipeline, including generative summarization
+For a more detailed examples of using Vectara with LangChain, see the following example notebooks:
+* [this notebook](/docs/integrations/vectorstores/vectara) shows how to use Vectara: with full RAG or just as a retriever.
 * [this notebook](/docs/integrations/retrievers/self_query/vectara_self_query) shows the self-query capability with Vectara.
-
-
+* [this notebook](/docs/integrations/providers/vectara/vectara_chat) shows how to build a chatbot with Langchain and Vectara

--- a/docs/docs/integrations/providers/vectara/vectara_chat.ipynb
+++ b/docs/docs/integrations/providers/vectara/vectara_chat.ipynb
@@ -5,7 +5,21 @@
   "id": "134a0785",
   "metadata": {},
   "source": [
-    "# Chat Over Documents with Vectara"
+    "# Vectara Chat\n",
+    "\n",
+    "[Vectara](https://vectara.com/) provides a Trusted Generative AI platform, allowing organizations to rapidly create a ChatGPT-like experience (an AI assistant) which is grounded in the data, documents, and knowledge that they have (technically, it is Retrieval-Augmented-Generation-as-a-service). \n",
+    "\n",
+    "Vectara serverless RAG-as-a-service provides all the components of RAG behind an easy-to-use API, including:\n",
+    "1. A way to extract text from files (PDF, PPT, DOCX, etc)\n",
+    "2. ML-based chunking that provides state of the art performance.\n",
+    "3. The [Boomerang](https://vectara.com/how-boomerang-takes-retrieval-augmented-generation-to-the-next-level-via-grounded-generation/) embeddings model.\n",
+    "4. Its own internal vector database where text chunks and embedding vectors are stored.\n",
+    "5. A query service that automatically encodes the query into embedding, and retrieves the most relevant text segments (including support for [Hybrid Search](https://docs.vectara.com/docs/api-reference/search-apis/lexical-matching) and [MMR](https://vectara.com/get-diverse-results-and-comprehensive-summaries-with-vectaras-mmr-reranker/))\n",
+    "7. An LLM to for creating a [generative summary](https://docs.vectara.com/docs/learn/grounded-generation/grounded-generation-overview), based on the retrieved documents (context), including citations.\n",
+    "\n",
+    "See the [Vectara API documentation](https://docs.vectara.com/docs/) for more information on how to use the API.\n",
+    "\n",
+    "This notebook shows how to use Vectara's [Chat](https://docs.vectara.com/docs/api-reference/chat-apis/chat-apis-overview) functionality."
   ]
  },
  {
@@ -13,19 +27,19 @@
   "id": "56372c5b",
   "metadata": {},
   "source": [
-    "# Setup\n",
+    "# Getting Started\n",
    "\n",
-    "You will need a Vectara account to use Vectara with LangChain. To get started, use the following steps:\n",
-    "1. [Sign up](https://www.vectara.com/integrations/langchain) for a Vectara account if you don't already have one. Once you have completed your sign up you will have a Vectara customer ID. You can find your customer ID by clicking on your name, on the top-right of the Vectara console window.\n",
+    "To get started, use the following steps:\n",
+    "1. If you don't already have one, [Sign up](https://www.vectara.com/integrations/langchain) for your free Vectara account. Once you have completed your sign up you will have a Vectara customer ID. You can find your customer ID by clicking on your name, on the top-right of the Vectara console window.\n",
    "2. Within your account you can create one or more corpora. Each corpus represents an area that stores text data upon ingest from input documents. To create a corpus, use the **\"Create Corpus\"** button. You then provide a name to your corpus as well as a description. Optionally you can define filtering attributes and apply some advanced options. If you click on your created corpus, you can see its name and corpus ID right on the top.\n",
-    "3. Next you'll need to create API keys to access the corpus. Click on the **\"Authorization\"** tab in the corpus view and then the **\"Create API Key\"** button. Give your key a name, and choose whether you want query only or query+index for your key. Click \"Create\" and you now have an active API key. Keep this key confidential. \n",
+    "3. Next you'll need to create API keys to access the corpus. Click on the **\"Access Control\"** tab in the corpus view and then the **\"Create API Key\"** button. Give your key a name, and choose whether you want query-only or query+index for your key. Click \"Create\" and you now have an active API key. Keep this key confidential. \n",
    "\n",
-    "To use LangChain with Vectara, you'll need to have these three values: customer ID, corpus ID and api_key.\n",
+    "To use LangChain with Vectara, you'll need to have these three values: `customer ID`, `corpus ID` and `api_key`.\n",
    "You can provide those to LangChain in two ways:\n",
    "\n",
    "1. Include in your environment these three variables: `VECTARA_CUSTOMER_ID`, `VECTARA_CORPUS_ID` and `VECTARA_API_KEY`.\n",
    "\n",
-    "> For example, you can set these variables using os.environ and getpass as follows:\n",
+    "   For example, you can set these variables using os.environ and getpass as follows:\n",
    "\n",
    "```python\n",
    "import os\n",
@@ -36,20 +50,21 @@
    "os.environ[\"VECTARA_API_KEY\"] = getpass.getpass(\"Vectara API Key:\")\n",
    "```\n",
    "\n",
-    "2. Add them to the Vectara vectorstore constructor:\n",
+    "2. Add them to the `Vectara` vectorstore constructor:\n",
    "\n",
    "```python\n",
-    "vectorstore = Vectara(\n",
+    "vectara = Vectara(\n",
    "                vectara_customer_id=vectara_customer_id,\n",
    "                vectara_corpus_id=vectara_corpus_id,\n",
    "                vectara_api_key=vectara_api_key\n",
    "            )\n",
-    "```"
+    "```\n",
+    "In this notebook we assume they are provided in the environment."
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 2,
+   "execution_count": 1,
   "id": "70c4e529",
   "metadata": {
    "tags": []
@@ -58,9 +73,16 @@
   "source": [
    "import os\n",
    "\n",
-    "from langchain.chains import ConversationalRetrievalChain\n",
+    "os.environ[\"VECTARA_API_KEY\"] = \"<YOUR_VECTARA_API_KEY>\"\n",
+    "os.environ[\"VECTARA_CORPUS_ID\"] = \"<YOUR_VECTARA_CORPUS_ID>\"\n",
+    "os.environ[\"VECTARA_CUSTOMER_ID\"] = \"<YOUR_VECTARA_CUSTOMER_ID>\"\n",
+    "\n",
    "from langchain_community.vectorstores import Vectara\n",
-    "from langchain_openai import OpenAI"
+    "from langchain_community.vectorstores.vectara import (\n",
+    "    RerankConfig,\n",
+    "    SummaryConfig,\n",
+    "    VectaraQueryConfig,\n",
+    ")"
   ]
  },
  {
@@ -68,62 +90,30 @@
   "id": "cdff94be",
   "metadata": {},
   "source": [
-    "Load in documents. You can replace this with a loader for whatever type of data you want"
+    "## Vectara Chat Explained\n",
+    "\n",
+    "In most uses of LangChain to create chatbots, one must integrate a special `memory` component that maintains the history of chat sessions and then uses that history to ensure the chatbot is aware of conversation history.\n",
+    "\n",
+    "With Vectara Chat - all of that is performed in the backend by Vectara automatically. You can look at the [Chat](https://docs.vectara.com/docs/api-reference/chat-apis/chat-apis-overview) documentation for the details, to learn more about the internals of how this is implemented, but with LangChain all you have to do is turn that feature on in the Vectara vectorstore.\n",
+    "\n",
+    "Let's see an example. First we load the SOTU document (remember, text extraction and chunking all occurs automatically on the Vectara platform):"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 3,
+   "execution_count": 2,
   "id": "01c46e92",
   "metadata": {
    "tags": []
   },
   "outputs": [],
   "source": [
-    "from langchain_community.document_loaders import TextLoader\n",
+    "from langchain.document_loaders import TextLoader\n",
    "\n",
    "loader = TextLoader(\"state_of_the_union.txt\")\n",
-    "documents = loader.load()"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "239475d2",
-   "metadata": {},
-   "source": [
-    "Since we're using Vectara, there's no need to chunk the documents, as that is done automatically in the Vectara platform backend. We just use `from_document()` to upload the text loaded from the file, and directly ingest it into Vectara:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 4,
-   "id": "a8930cf7",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [],
-   "source": [
-    "vectara = Vectara.from_documents(documents, embedding=None)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "898b574b",
-   "metadata": {},
-   "source": [
-    "We can now create a memory object, which is neccessary to track the inputs/outputs and hold a conversation."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 5,
-   "id": "af803fee",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "from langchain.memory import ConversationBufferMemory\n",
+    "documents = loader.load()\n",
    "\n",
-    "memory = ConversationBufferMemory(memory_key=\"chat_history\", return_messages=True)"
+    "vectara = Vectara.from_documents(documents, embedding=None)"
   ]
  },
  {
@@ -131,139 +121,25 @@
   "id": "3c96b118",
   "metadata": {},
   "source": [
-    "We now initialize the `ConversationalRetrievalChain`:"
+    "And now we create a Chat Runnable using the `as_chat` method:"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 6,
-   "id": "7b4110f3",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "[Document(page_content='Justice Breyer, thank you for your service. One of the most serious constitutional responsibilities a President has is nominating someone to serve on the United States Supreme Court. And I did that 4 days ago, when I nominated Circuit Court of Appeals Judge Ketanji Brown Jackson. One of our nation’s top legal minds, who will continue Justice Breyer’s legacy of excellence. A former top litigator in private practice.', metadata={'source': 'langchain', 'lang': 'eng', 'offset': '29486', 'len': '97'}), Document(page_content='Groups of citizens blocking tanks with their bodies. Everyone from students to retirees teachers turned soldiers defending their homeland. In this struggle as President Zelenskyy said in his speech to the European Parliament “Light will win over darkness.” The Ukrainian Ambassador to the United States is here tonight. Let each of us here tonight in this Chamber send an unmistakable signal to Ukraine and to the world.', metadata={'source': 'langchain', 'lang': 'eng', 'offset': '1083', 'len': '117'}), Document(page_content='All told, we created 369,000 new manufacturing jobs in America just last year. Powered by people I’ve met like JoJo Burgess, from generations of union steelworkers from Pittsburgh, who’s here with us tonight. As Ohio Senator Sherrod Brown says, “It’s time to bury the label “Rust Belt.” It’s time. \\n\\nBut with all the bright spots in our economy, record job growth and higher wages, too many families are struggling to keep up with the bills. Inflation is robbing them of the gains they might otherwise feel.', metadata={'source': 'langchain', 'lang': 'eng', 'offset': '14257', 'len': '77'}), Document(page_content='This is personal to me and Jill, to Kamala, and to so many of you. Cancer is the #2 cause of death in America–second only to heart disease. Last month, I announced our plan to supercharge  \\nthe Cancer Moonshot that President Obama asked me to lead six years ago. Our goal is to cut the cancer death rate by at least 50% over the next 25 years, turn more cancers from death sentences into treatable diseases. More support for patients and families.', metadata={'source': 'langchain', 'lang': 'eng', 'offset': '36196', 'len': '122'}), Document(page_content='Six days ago, Russia’s Vladimir Putin sought to shake the foundations of the free world thinking he could make it bend to his menacing ways. But he badly miscalculated. He thought he could roll into Ukraine and the world would roll over. Instead he met a wall of strength he never imagined. He met the Ukrainian people.', metadata={'source': 'langchain', 'lang': 'eng', 'offset': '664', 'len': '68'}), Document(page_content='I understand. \\n\\nI remember when my Dad had to leave our home in Scranton, Pennsylvania to find work. I grew up in a family where if the price of food went up, you felt it. That’s why one of the first things I did as President was fight to pass the American Rescue Plan. Because people were hurting. We needed to act, and we did.', metadata={'source': 'langchain', 'lang': 'eng', 'offset': '8042', 'len': '97'}), Document(page_content='He rejected repeated efforts at diplomacy. He thought the West and NATO wouldn’t respond. And he thought he could divide us at home. We were ready.  Here is what we did. We prepared extensively and carefully.', metadata={'source': 'langchain', 'lang': 'eng', 'offset': '2100', 'len': '42'}), Document(page_content='He thought he could roll into Ukraine and the world would roll over. Instead he met a wall of strength he never imagined. He met the Ukrainian people. From President Zelenskyy to every Ukrainian, their fearlessness, their courage, their determination, inspires the world. Groups of citizens blocking tanks with their bodies.', metadata={'source': 'langchain', 'lang': 'eng', 'offset': '788', 'len': '28'}), Document(page_content='Putin’s latest attack on Ukraine was premeditated and unprovoked. He rejected repeated efforts at diplomacy. He thought the West and NATO wouldn’t respond. And he thought he could divide us at home. We were ready.  Here is what we did.', metadata={'source': 'langchain', 'lang': 'eng', 'offset': '2053', 'len': '46'}), Document(page_content='A unity agenda for the nation. We can do this. \\n\\nMy fellow Americans—tonight , we have gathered in a sacred space—the citadel of our democracy. In this Capitol, generation after generation, Americans have debated great questions amid great strife, and have done great things. We have fought for freedom, expanded liberty, defeated totalitarianism and terror. And built the strongest, freest, and most prosperous nation the world has ever known.', metadata={'source': 'langchain', 'lang': 'eng', 'offset': '36968', 'len': '131'})]\n"
-     ]
-    }
-   ],
-   "source": [
-    "openai_api_key = os.environ[\"OPENAI_API_KEY\"]\n",
-    "llm = OpenAI(openai_api_key=openai_api_key, temperature=0)\n",
-    "retriever = vectara.as_retriever()\n",
-    "d = retriever.invoke(\"What did the president say about Ketanji Brown Jackson\", k=2)\n",
-    "print(d)"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 7,
-   "id": "44ed803e",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "bot = ConversationalRetrievalChain.from_llm(\n",
-    "    llm, retriever, memory=memory, verbose=False\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "5b6deb16",
-   "metadata": {},
-   "source": [
-    "And can have a multi-turn conversation with out new bot:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 8,
-   "id": "e8ce4fe9",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "query = \"What did the president say about Ketanji Brown Jackson\"\n",
-    "result = bot.invoke({\"question\": query})"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 9,
-   "id": "4c79862b",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "\" The president said that Ketanji Brown Jackson is one of the nation's top legal minds and a former top litigator in private practice, and that she will continue Justice Breyer's legacy of excellence.\""
-      ]
-     },
-     "execution_count": 9,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "result[\"answer\"]"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 10,
-   "id": "c697d9d1",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "query = \"Did he mention who she suceeded\"\n",
-    "result = bot.invoke({\"question\": query})"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 11,
-   "id": "ba0678f3",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "' Ketanji Brown Jackson succeeded Justice Breyer on the United States Supreme Court.'"
-      ]
-     },
-     "execution_count": 11,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "result[\"answer\"]"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "b3308b01-5300-4999-8cd3-22f16dae757e",
-   "metadata": {},
-   "source": [
-    "## Pass in chat history\n",
-    "\n",
-    "In the above example, we used a Memory object to track chat history. We can also just pass it in explicitly. In order to do this, we need to initialize a chain without any memory object."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 12,
+   "execution_count": 3,
   "id": "1b41a10b-bf68-4689-8f00-9aed7675e2ab",
   "metadata": {
    "tags": []
   },
   "outputs": [],
   "source": [
-    "bot = ConversationalRetrievalChain.from_llm(\n",
-    "    OpenAI(temperature=0), vectara.as_retriever()\n",
-    ")"
+    "summary_config = SummaryConfig(is_enabled=True, max_results=7, response_lang=\"eng\")\n",
+    "rerank_config = RerankConfig(reranker=\"mmr\", rerank_k=50, mmr_diversity_bias=0.2)\n",
+    "config = VectaraQueryConfig(\n",
+    "    k=10, lambda_val=0.005, rerank_config=rerank_config, summary_config=summary_config\n",
+    ")\n",
+    "\n",
+    "bot = vectara.as_chat(config)"
   ]
  },
  {
@@ -276,39 +152,25 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 13,
+   "execution_count": 4,
   "id": "bc672290-8a8b-4828-a90c-f1bbdd6b3920",
   "metadata": {
    "tags": []
   },
-   "outputs": [],
-   "source": [
-    "chat_history = []\n",
-    "query = \"What did the president say about Ketanji Brown Jackson\"\n",
-    "result = bot.invoke({\"question\": query, \"chat_history\": chat_history})"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 14,
-   "id": "6b62d758-c069-4062-88f0-21e7ea4710bf",
-   "metadata": {
-    "tags": []
-   },
   "outputs": [
    {
     "data": {
      "text/plain": [
-       "\" The president said that Ketanji Brown Jackson is one of the nation's top legal minds and a former top litigator in private practice, and that she will continue Justice Breyer's legacy of excellence.\""
+       "'The President expressed gratitude to Justice Breyer and highlighted the significance of nominating Ketanji Brown Jackson to the Supreme Court, praising her legal expertise and commitment to upholding excellence [1]. The President also reassured the public about the situation with gas prices and the conflict in Ukraine, emphasizing unity with allies and the belief that the world will emerge stronger from these challenges [2][4]. Additionally, the President shared personal experiences related to economic struggles and the importance of passing the American Rescue Plan to support those in need [3]. The focus was also on job creation and economic growth, acknowledging the impact of inflation on families [5]. While addressing cancer as a significant issue, the President discussed plans to enhance cancer research and support for patients and families [7].'"
      ]
     },
-     "execution_count": 14,
+     "execution_count": 4,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
-    "result[\"answer\"]"
+    "bot.invoke(\"What did the president say about Ketanji Brown Jackson?\")[\"answer\"]"
   ]
  },
  {
@@ -321,256 +183,25 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 15,
+   "execution_count": 5,
   "id": "9c95460b-7116-4155-a9d2-c0fb027ee592",
   "metadata": {
    "tags": []
   },
-   "outputs": [],
-   "source": [
-    "chat_history = [(query, result[\"answer\"])]\n",
-    "query = \"Did he mention who she suceeded\"\n",
-    "result = bot.invoke({\"question\": query, \"chat_history\": chat_history})"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 16,
-   "id": "698ac00c-cadc-407f-9423-226b2d9258d0",
-   "metadata": {
-    "tags": []
-   },
   "outputs": [
    {
     "data": {
      "text/plain": [
-       "' Ketanji Brown Jackson succeeded Justice Breyer on the United States Supreme Court.'"
+       "\"In his remarks, the President specified that Ketanji Brown Jackson is succeeding Justice Breyer on the United States Supreme Court[1]. The President praised Jackson as a top legal mind who will continue Justice Breyer's legacy of excellence. The nomination of Jackson was highlighted as a significant constitutional responsibility of the President[1]. The President emphasized the importance of this nomination and the qualities that Jackson brings to the role. The focus was on the transition from Justice Breyer to Judge Ketanji Brown Jackson on the Supreme Court[1].\""
      ]
     },
-     "execution_count": 16,
+     "execution_count": 5,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
-    "result[\"answer\"]"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "0eaadf0f",
-   "metadata": {},
-   "source": [
-    "## Return Source Documents\n",
-    "You can also easily return source documents from the ConversationalRetrievalChain. This is useful for when you want to inspect what documents were returned."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 17,
-   "id": "562769c6",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [],
-   "source": [
-    "bot = ConversationalRetrievalChain.from_llm(\n",
-    "    llm, vectara.as_retriever(), return_source_documents=True\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 18,
-   "id": "ea478300",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [],
-   "source": [
-    "chat_history = []\n",
-    "query = \"What did the president say about Ketanji Brown Jackson\"\n",
-    "result = bot.invoke({\"question\": query, \"chat_history\": chat_history})"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 19,
-   "id": "4cb75b4e",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "Document(page_content='Justice Breyer, thank you for your service. One of the most serious constitutional responsibilities a President has is nominating someone to serve on the United States Supreme Court. And I did that 4 days ago, when I nominated Circuit Court of Appeals Judge Ketanji Brown Jackson. One of our nation’s top legal minds, who will continue Justice Breyer’s legacy of excellence. A former top litigator in private practice.', metadata={'source': 'langchain', 'lang': 'eng', 'offset': '29486', 'len': '97'})"
-      ]
-     },
-     "execution_count": 19,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "result[\"source_documents\"][0]"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "99b96dae",
-   "metadata": {},
-   "source": [
-    "## ConversationalRetrievalChain with `map_reduce`\n",
-    "LangChain supports different types of ways to combine document chains with the ConversationalRetrievalChain chain."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 20,
-   "id": "e53a9d66",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [],
-   "source": [
-    "from langchain.chains import LLMChain\n",
-    "from langchain.chains.conversational_retrieval.prompts import CONDENSE_QUESTION_PROMPT\n",
-    "from langchain.chains.question_answering import load_qa_chain"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 21,
-   "id": "bf205e35",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [],
-   "source": [
-    "question_generator = LLMChain(llm=llm, prompt=CONDENSE_QUESTION_PROMPT)\n",
-    "doc_chain = load_qa_chain(llm, chain_type=\"map_reduce\")\n",
-    "\n",
-    "chain = ConversationalRetrievalChain(\n",
-    "    retriever=vectara.as_retriever(),\n",
-    "    question_generator=question_generator,\n",
-    "    combine_docs_chain=doc_chain,\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 22,
-   "id": "78155887",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [],
-   "source": [
-    "chat_history = []\n",
-    "query = \"What did the president say about Ketanji Brown Jackson\"\n",
-    "result = chain({\"question\": query, \"chat_history\": chat_history})"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 23,
-   "id": "e54b5fa2",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "\" The president said that he nominated Circuit Court of Appeals Judge Ketanji Brown Jackson, who is one of the nation's top legal minds and a former top litigator in private practice.\""
-      ]
-     },
-     "execution_count": 23,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "result[\"answer\"]"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "a2fe6b14",
-   "metadata": {},
-   "source": [
-    "## ConversationalRetrievalChain with Question Answering with sources\n",
-    "\n",
-    "You can also use this chain with the question answering with sources chain."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 24,
-   "id": "d1058fd2",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [],
-   "source": [
-    "from langchain.chains.qa_with_sources import load_qa_with_sources_chain"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 25,
-   "id": "a6594482",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [],
-   "source": [
-    "question_generator = LLMChain(llm=llm, prompt=CONDENSE_QUESTION_PROMPT)\n",
-    "doc_chain = load_qa_with_sources_chain(llm, chain_type=\"map_reduce\")\n",
-    "\n",
-    "chain = ConversationalRetrievalChain(\n",
-    "    retriever=vectara.as_retriever(),\n",
-    "    question_generator=question_generator,\n",
-    "    combine_docs_chain=doc_chain,\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 26,
-   "id": "e2badd21",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [],
-   "source": [
-    "chat_history = []\n",
-    "query = \"What did the president say about Ketanji Brown Jackson\"\n",
-    "result = chain({\"question\": query, \"chat_history\": chat_history})"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 27,
-   "id": "edb31fe5",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "\" The president said that Ketanji Brown Jackson is one of the nation's top legal minds and a former top litigator in private practice.\\nSOURCES: langchain\""
-      ]
-     },
-     "execution_count": 27,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "result[\"answer\"]"
+    "bot.invoke(\"Did he mention who she suceeded?\")[\"answer\"]"
   ]
  },
  {
@@ -578,157 +209,40 @@
   "id": "2324cdc6-98bf-4708-b8cd-02a98b1e5b67",
   "metadata": {},
   "source": [
-    "## ConversationalRetrievalChain with streaming to `stdout`\n",
+    "## Chat with streaming\n",
    "\n",
-    "Output from the chain will be streamed to `stdout` token by token in this example."
+    "Of course the chatbot interface also supports streaming.\n",
+    "Instead of the `invoke` method you simply use `stream`:"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 28,
-   "id": "2efacec3-2690-4b05-8de3-a32fd2ac3911",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [],
-   "source": [
-    "from langchain.chains.conversational_retrieval.prompts import (\n",
-    "    CONDENSE_QUESTION_PROMPT,\n",
-    "    QA_PROMPT,\n",
-    ")\n",
-    "from langchain.chains.llm import LLMChain\n",
-    "from langchain.chains.question_answering import load_qa_chain\n",
-    "from langchain_core.callbacks import StreamingStdOutCallbackHandler\n",
-    "\n",
-    "# Construct a ConversationalRetrievalChain with a streaming llm for combine docs\n",
-    "# and a separate, non-streaming llm for question generation\n",
-    "llm = OpenAI(temperature=0, openai_api_key=openai_api_key)\n",
-    "streaming_llm = OpenAI(\n",
-    "    streaming=True,\n",
-    "    callbacks=[StreamingStdOutCallbackHandler()],\n",
-    "    temperature=0,\n",
-    "    openai_api_key=openai_api_key,\n",
-    ")\n",
-    "\n",
-    "question_generator = LLMChain(llm=llm, prompt=CONDENSE_QUESTION_PROMPT)\n",
-    "doc_chain = load_qa_chain(streaming_llm, chain_type=\"stuff\", prompt=QA_PROMPT)\n",
-    "\n",
-    "bot = ConversationalRetrievalChain(\n",
-    "    retriever=vectara.as_retriever(),\n",
-    "    combine_docs_chain=doc_chain,\n",
-    "    question_generator=question_generator,\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 29,
-   "id": "fd6d43f4-7428-44a4-81bc-26fe88a98762",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      " The president said that Ketanji Brown Jackson is one of the nation's top legal minds and a former top litigator in private practice, and that she will continue Justice Breyer's legacy of excellence."
-     ]
-    }
-   ],
-   "source": [
-    "chat_history = []\n",
-    "query = \"What did the president say about Ketanji Brown Jackson\"\n",
-    "result = bot.invoke({\"question\": query, \"chat_history\": chat_history})"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 30,
-   "id": "5ab38978-f3e8-4fa7-808c-c79dec48379a",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      " Ketanji Brown Jackson succeeded Justice Breyer on the United States Supreme Court."
-     ]
-    }
-   ],
-   "source": [
-    "chat_history = [(query, result[\"answer\"])]\n",
-    "query = \"Did he mention who she suceeded\"\n",
-    "result = bot.invoke({\"question\": query, \"chat_history\": chat_history})"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "f793d56b",
-   "metadata": {},
-   "source": [
-    "## get_chat_history Function\n",
-    "You can also specify a `get_chat_history` function, which can be used to format the chat_history string."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 31,
-   "id": "a7ba9d8c",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [],
-   "source": [
-    "def get_chat_history(inputs) -> str:\n",
-    "    res = []\n",
-    "    for human, ai in inputs:\n",
-    "        res.append(f\"Human:{human}\\nAI:{ai}\")\n",
-    "    return \"\\n\".join(res)\n",
-    "\n",
-    "\n",
-    "bot = ConversationalRetrievalChain.from_llm(\n",
-    "    llm, vectara.as_retriever(), get_chat_history=get_chat_history\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 32,
-   "id": "a3e33c0d",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [],
-   "source": [
-    "chat_history = []\n",
-    "query = \"What did the president say about Ketanji Brown Jackson\"\n",
-    "result = bot.invoke({\"question\": query, \"chat_history\": chat_history})"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 33,
+   "execution_count": 6,
   "id": "936dc62f",
   "metadata": {
    "tags": []
   },
   "outputs": [
    {
-     "data": {
-      "text/plain": [
-       "\" The president said that Ketanji Brown Jackson is one of the nation's top legal minds and a former top litigator in private practice, and that she will continue Justice Breyer's legacy of excellence.\""
-      ]
-     },
-     "execution_count": 33,
-     "metadata": {},
-     "output_type": "execute_result"
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "Judge Ketanji Brown Jackson is a nominee for the United States Supreme Court, known for her legal expertise and experience as a former litigator. She is praised for her potential to continue the legacy of excellence on the Court[1]. While the search results provide information on various topics like innovation, economic growth, and healthcare initiatives, they do not directly address Judge Ketanji Brown Jackson's specific accomplishments. Therefore, I do not have enough information to answer this question."
+     ]
    }
   ],
   "source": [
-    "result[\"answer\"]"
+    "output = {}\n",
+    "curr_key = None\n",
+    "for chunk in bot.stream(\"what about her accopmlishments?\"):\n",
+    "    for key in chunk:\n",
+    "        if key not in output:\n",
+    "            output[key] = chunk[key]\n",
+    "        else:\n",
+    "            output[key] += chunk[key]\n",
+    "        if key == \"answer\":\n",
+    "            print(chunk[key], end=\"\", flush=True)\n",
+    "        curr_key = key"
   ]
  }
 ],
@@ -748,7 +262,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
-   "version": "3.11.6"
+   "version": "3.11.8"
  }
 },
 "nbformat": 4,
--- a/docs/docs/integrations/providers/vectara/vectara_summary.ipynb
+++ b/docs/docs/integrations/providers/vectara/vectara_summary.ipynb
@@ -1,311 +0,0 @@
-{
- "cells": [
-  {
-   "cell_type": "markdown",
-   "id": "559f8e0e",
-   "metadata": {},
-   "source": [
-    "# Vectara\n",
-    "\n",
-    ">[Vectara](https://vectara.com/) is the trusted GenAI platform that provides an easy-to-use API for document indexing and querying. \n",
-    "\n",
-    "Vectara provides an end-to-end managed service for Retrieval Augmented Generation or [RAG](https://vectara.com/grounded-generation/), which includes:\n",
-    "\n",
-    "1. A way to extract text from document files and chunk them into sentences.\n",
-    "\n",
-    "2. The state-of-the-art [Boomerang](https://vectara.com/how-boomerang-takes-retrieval-augmented-generation-to-the-next-level-via-grounded-generation/) embeddings model. Each text chunk is encoded into a vector embedding using Boomerang, and stored in the Vectara internal knowledge (vector+text) store\n",
-    "\n",
-    "3. A query service that automatically encodes the query into embedding, and retrieves the most relevant text segments (including support for [Hybrid Search](https://docs.vectara.com/docs/api-reference/search-apis/lexical-matching) and [MMR](https://vectara.com/get-diverse-results-and-comprehensive-summaries-with-vectaras-mmr-reranker/))\n",
-    "\n",
-    "4. An option to create [generative summary](https://docs.vectara.com/docs/learn/grounded-generation/grounded-generation-overview), based on the retrieved documents, including citations.\n",
-    "\n",
-    "See the [Vectara API documentation](https://docs.vectara.com/docs/) for more information on how to use the API.\n",
-    "\n",
-    "This notebook shows how to use functionality related to the `Vectara`'s integration with langchain.\n",
-    "Specificaly we will demonstrate how to use chaining with [LangChain's Expression Language](/docs/concepts#langchain-expression-language) and using Vectara's integrated summarization capability."
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "e97dcf11",
-   "metadata": {},
-   "source": [
-    "# Setup\n",
-    "\n",
-    "You will need a Vectara account to use Vectara with LangChain. To get started, use the following steps:\n",
-    "\n",
-    "1. [Sign up](https://www.vectara.com/integrations/langchain) for a Vectara account if you don't already have one. Once you have completed your sign up you will have a Vectara customer ID. You can find your customer ID by clicking on your name, on the top-right of the Vectara console window.\n",
-    "\n",
-    "2. Within your account you can create one or more corpora. Each corpus represents an area that stores text data upon ingest from input documents. To create a corpus, use the **\"Create Corpus\"** button. You then provide a name to your corpus as well as a description. Optionally you can define filtering attributes and apply some advanced options. If you click on your created corpus, you can see its name and corpus ID right on the top.\n",
-    "\n",
-    "3. Next you'll need to create API keys to access the corpus. Click on the **\"Authorization\"** tab in the corpus view and then the **\"Create API Key\"** button. Give your key a name, and choose whether you want query only or query+index for your key. Click \"Create\" and you now have an active API key. Keep this key confidential. \n",
-    "\n",
-    "To use LangChain with Vectara, you'll need to have these three values: customer ID, corpus ID and api_key.\n",
-    "You can provide those to LangChain in two ways:\n",
-    "\n",
-    "1. Include in your environment these three variables: `VECTARA_CUSTOMER_ID`, `VECTARA_CORPUS_ID` and `VECTARA_API_KEY`.\n",
-    "\n",
-    "> For example, you can set these variables using os.environ and getpass as follows:\n",
-    "\n",
-    "```python\n",
-    "import os\n",
-    "import getpass\n",
-    "\n",
-    "os.environ[\"VECTARA_CUSTOMER_ID\"] = getpass.getpass(\"Vectara Customer ID:\")\n",
-    "os.environ[\"VECTARA_CORPUS_ID\"] = getpass.getpass(\"Vectara Corpus ID:\")\n",
-    "os.environ[\"VECTARA_API_KEY\"] = getpass.getpass(\"Vectara API Key:\")\n",
-    "```\n",
-    "\n",
-    "2. Add them to the Vectara vectorstore constructor:\n",
-    "\n",
-    "```python\n",
-    "vectorstore = Vectara(\n",
-    "                vectara_customer_id=vectara_customer_id,\n",
-    "                vectara_corpus_id=vectara_corpus_id,\n",
-    "                vectara_api_key=vectara_api_key\n",
-    "            )\n",
-    "```"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 1,
-   "id": "aac7a9a6",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "from langchain_community.embeddings import FakeEmbeddings\n",
-    "from langchain_community.vectorstores import Vectara\n",
-    "from langchain_core.output_parsers import StrOutputParser\n",
-    "from langchain_core.prompts import ChatPromptTemplate\n",
-    "from langchain_core.runnables import RunnableLambda, RunnablePassthrough"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "875ffb7e",
-   "metadata": {},
-   "source": [
-    "First we load the state-of-the-union text into Vectara. Note that we use the `from_files` interface which does not require any local processing or chunking - Vectara receives the file content and performs all the necessary pre-processing, chunking and embedding of the file into its knowledge store."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 2,
-   "id": "be0a4973",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "vectara = Vectara.from_files([\"state_of_the_union.txt\"])"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "22a6b953",
-   "metadata": {},
-   "source": [
-    "We now create a Vectara retriever and specify that:\n",
-    "* It should return only the 3 top Document matches\n",
-    "* For summary, it should use the top 5 results and respond in English"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 3,
-   "id": "19cd2f86",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "summary_config = {\"is_enabled\": True, \"max_results\": 5, \"response_lang\": \"eng\"}\n",
-    "retriever = vectara.as_retriever(\n",
-    "    search_kwargs={\"k\": 3, \"summary_config\": summary_config}\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "c49284ed",
-   "metadata": {},
-   "source": [
-    "When using summarization with Vectara, the retriever responds with a list of `Document` objects:\n",
-    "1. The first `k` documents are the ones that match the query (as we are used to with a standard vector store)\n",
-    "2. With summary enabled, an additional `Document` object is apended, which includes the summary text. This Document has the metadata field `summary` set as True.\n",
-    "\n",
-    "Let's define two utility functions to split those out:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 4,
-   "id": "e5100654",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "def get_sources(documents):\n",
-    "    return documents[:-1]\n",
-    "\n",
-    "\n",
-    "def get_summary(documents):\n",
-    "    return documents[-1].page_content\n",
-    "\n",
-    "\n",
-    "query_str = \"what did Biden say?\""
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "f2a74368",
-   "metadata": {},
-   "source": [
-    "Now we can try a summary response for the query:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 5,
-   "id": "ee4759c4",
-   "metadata": {
-    "scrolled": false
-   },
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "'The returned results did not contain sufficient information to be summarized into a useful answer for your query. Please try a different search or restate your query differently.'"
-      ]
-     },
-     "execution_count": 5,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "(retriever | get_summary).invoke(query_str)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "dd7c4593",
-   "metadata": {},
-   "source": [
-    "And if we would like to see the sources retrieved from Vectara that were used in this summary (the citations):"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 6,
-   "id": "0eb66034",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[Document(page_content='When they came home, many of the world’s fittest and best trained warriors were never the same. Dizziness. \\n\\nA cancer that would put them in a flag-draped coffin. I know. \\n\\nOne of those soldiers was my son Major Beau Biden. We don’t know for sure if a burn pit was the cause of his brain cancer, or the diseases of so many of our troops. But I’m committed to finding out everything we can.', metadata={'lang': 'eng', 'section': '1', 'offset': '34652', 'len': '60', 'X-TIKA:Parsed-By': 'org.apache.tika.parser.csv.TextAndCSVParser', 'Content-Encoding': 'UTF-8', 'Content-Type': 'text/plain; charset=UTF-8', 'source': 'vectara'}),\n",
-       " Document(page_content='The U.S. Department of Justice is assembling a dedicated task force to go after the crimes of Russian oligarchs. We are joining with our European allies to find and seize your yachts your luxury apartments your private jets. We are coming for your ill-begotten gains. And tonight I am announcing that we will join our allies in closing off American air space to all Russian flights – further isolating Russia – and adding an additional squeeze –on their economy. The Ruble has lost 30% of its value.', metadata={'lang': 'eng', 'section': '1', 'offset': '3807', 'len': '42', 'X-TIKA:Parsed-By': 'org.apache.tika.parser.csv.TextAndCSVParser', 'Content-Encoding': 'UTF-8', 'Content-Type': 'text/plain; charset=UTF-8', 'source': 'vectara'}),\n",
-       " Document(page_content='He rejected repeated efforts at diplomacy. He thought the West and NATO wouldn’t respond. And he thought he could divide us at home. We were ready.  Here is what we did. We prepared extensively and carefully.', metadata={'lang': 'eng', 'section': '1', 'offset': '2100', 'len': '42', 'X-TIKA:Parsed-By': 'org.apache.tika.parser.csv.TextAndCSVParser', 'Content-Encoding': 'UTF-8', 'Content-Type': 'text/plain; charset=UTF-8', 'source': 'vectara'})]"
-      ]
-     },
-     "execution_count": 6,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "(retriever | get_sources).invoke(query_str)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "8f16bf8d",
-   "metadata": {},
-   "source": [
-    "Vectara's \"RAG as a service\" does a lot of the heavy lifting in creating question answering or chatbot chains. The integration with LangChain provides the option to use additional capabilities such as query pre-processing  like `SelfQueryRetriever` or `MultiQueryRetriever`. Let's look at an example of using the [MultiQueryRetriever](/docs/how_to/MultiQueryRetriever).\n",
-    "\n",
-    "Since MQR uses an LLM we have to set that up - here we choose `ChatOpenAI`:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 7,
-   "id": "e14325b9",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "\"President Biden has made several notable quotes and comments. He expressed a commitment to investigate the potential impact of burn pits on soldiers' health, referencing his son's brain cancer [1]. He emphasized the importance of unity among Americans, urging us to see each other as fellow citizens rather than enemies [2]. Biden also highlighted the need for schools to use funds from the American Rescue Plan to hire teachers and address learning loss, while encouraging community involvement in supporting education [3].\""
-      ]
-     },
-     "execution_count": 7,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "from langchain.retrievers.multi_query import MultiQueryRetriever\n",
-    "from langchain_openai import ChatOpenAI\n",
-    "\n",
-    "llm = ChatOpenAI(temperature=0)\n",
-    "mqr = MultiQueryRetriever.from_llm(retriever=retriever, llm=llm)\n",
-    "\n",
-    "(mqr | get_summary).invoke(query_str)"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 8,
-   "id": "fa14f923",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[Document(page_content='When they came home, many of the world’s fittest and best trained warriors were never the same. Dizziness. \\n\\nA cancer that would put them in a flag-draped coffin. I know. \\n\\nOne of those soldiers was my son Major Beau Biden. We don’t know for sure if a burn pit was the cause of his brain cancer, or the diseases of so many of our troops. But I’m committed to finding out everything we can.', metadata={'lang': 'eng', 'section': '1', 'offset': '34652', 'len': '60', 'X-TIKA:Parsed-By': 'org.apache.tika.parser.csv.TextAndCSVParser', 'Content-Encoding': 'UTF-8', 'Content-Type': 'text/plain; charset=UTF-8', 'source': 'vectara'}),\n",
-       " Document(page_content='The U.S. Department of Justice is assembling a dedicated task force to go after the crimes of Russian oligarchs. We are joining with our European allies to find and seize your yachts your luxury apartments your private jets. We are coming for your ill-begotten gains. And tonight I am announcing that we will join our allies in closing off American air space to all Russian flights – further isolating Russia – and adding an additional squeeze –on their economy. The Ruble has lost 30% of its value.', metadata={'lang': 'eng', 'section': '1', 'offset': '3807', 'len': '42', 'X-TIKA:Parsed-By': 'org.apache.tika.parser.csv.TextAndCSVParser', 'Content-Encoding': 'UTF-8', 'Content-Type': 'text/plain; charset=UTF-8', 'source': 'vectara'}),\n",
-       " Document(page_content='And, if Congress provides the funds we need, we’ll have new stockpiles of tests, masks, and pills ready if needed. I cannot promise a new variant won’t come. But I can promise you we’ll do everything within our power to be ready if it does. Third – we can end the shutdown of schools and businesses. We have the tools we need.', metadata={'lang': 'eng', 'section': '1', 'offset': '24753', 'len': '82', 'X-TIKA:Parsed-By': 'org.apache.tika.parser.csv.TextAndCSVParser', 'Content-Encoding': 'UTF-8', 'Content-Type': 'text/plain; charset=UTF-8', 'source': 'vectara'}),\n",
-       " Document(page_content='The returned results did not contain sufficient information to be summarized into a useful answer for your query. Please try a different search or restate your query differently.', metadata={'summary': True}),\n",
-       " Document(page_content='Danielle says Heath was a fighter to the very end. He didn’t know how to stop fighting, and neither did she. Through her pain she found purpose to demand we do better. Tonight, Danielle—we are. The VA is pioneering new ways of linking toxic exposures to diseases, already helping more veterans get benefits.', metadata={'lang': 'eng', 'section': '1', 'offset': '35502', 'len': '58', 'X-TIKA:Parsed-By': 'org.apache.tika.parser.csv.TextAndCSVParser', 'Content-Encoding': 'UTF-8', 'Content-Type': 'text/plain; charset=UTF-8', 'source': 'vectara'}),\n",
-       " Document(page_content='Let’s stop seeing each other as enemies, and start seeing each other for who we really are: Fellow Americans. We can’t change how divided we’ve been. But we can change how we move forward—on COVID-19 and other issues we must face together. I recently visited the New York City Police Department days after the funerals of Officer Wilbert Mora and his partner, Officer Jason Rivera. They were responding to a 9-1-1 call when a man shot and killed them with a stolen gun.', metadata={'lang': 'eng', 'section': '1', 'offset': '26312', 'len': '89', 'X-TIKA:Parsed-By': 'org.apache.tika.parser.csv.TextAndCSVParser', 'Content-Encoding': 'UTF-8', 'Content-Type': 'text/plain; charset=UTF-8', 'source': 'vectara'}),\n",
-       " Document(page_content='The American Rescue Plan gave schools money to hire teachers and help students make up for lost learning. I urge every parent to make sure your school does just that. And we can all play a part—sign up to be a tutor or a mentor. Children were also struggling before the pandemic. Bullying, violence, trauma, and the harms of social media.', metadata={'lang': 'eng', 'section': '1', 'offset': '33227', 'len': '61', 'X-TIKA:Parsed-By': 'org.apache.tika.parser.csv.TextAndCSVParser', 'Content-Encoding': 'UTF-8', 'Content-Type': 'text/plain; charset=UTF-8', 'source': 'vectara'})]"
-      ]
-     },
-     "execution_count": 8,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "(mqr | get_sources).invoke(query_str)"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "16853820",
-   "metadata": {},
-   "outputs": [],
-   "source": []
-  }
- ],
- "metadata": {
-  "kernelspec": {
-   "display_name": "Python 3 (ipykernel)",
-   "language": "python",
-   "name": "python3"
-  },
-  "language_info": {
-   "codemirror_mode": {
-    "name": "ipython",
-    "version": 3
-   },
-   "file_extension": ".py",
-   "mimetype": "text/x-python",
-   "name": "python",
-   "nbconvert_exporter": "python",
-   "pygments_lexer": "ipython3",
-   "version": "3.10.9"
-  }
- },
- "nbformat": 4,
- "nbformat_minor": 5
-}
--- a/docs/docs/integrations/retrievers/kinetica.ipynb
+++ b/docs/docs/integrations/retrievers/kinetica.ipynb
@@ -22,7 +22,7 @@
   "outputs": [],
   "source": [
    "# Please ensure that this connector is installed in your working environment.\n",
-    "%pip install gpudb==7.2.0.1"
+    "%pip install gpudb==7.2.0.9"
   ]
  },
  {
--- a/docs/docs/integrations/retrievers/self_query/vectara_self_query.ipynb
+++ b/docs/docs/integrations/retrievers/self_query/vectara_self_query.ipynb
@@ -5,15 +5,17 @@
   "id": "13afcae7",
   "metadata": {},
   "source": [
-    "# Vectara \n",
+    "# Vectara self-querying \n",
    "\n",
-    ">[Vectara](https://vectara.com/) is the trusted GenAI platform that provides an easy-to-use API for document indexing and querying. \n",
-    ">\n",
-    ">`Vectara` provides an end-to-end managed service for `Retrieval Augmented Generation` or [RAG](https://vectara.com/grounded-generation/), which includes:\n",
-    ">1. A way to `extract text` from document files and `chunk` them into sentences.\n",
-    ">2. The state-of-the-art [Boomerang](https://vectara.com/how-boomerang-takes-retrieval-augmented-generation-to-the-next-level-via-grounded-generation/) embeddings model. Each text chunk is encoded into a vector embedding using `Boomerang`, and stored in the Vectara internal knowledge (vector+text) store\n",
-    ">3. A query service that automatically encodes the query into embedding, and retrieves the most relevant text segments (including support for [Hybrid Search](https://docs.vectara.com/docs/api-reference/search-apis/lexical-matching) and [MMR](https://vectara.com/get-diverse-results-and-comprehensive-summaries-with-vectaras-mmr-reranker/))\n",
-    ">4. An option to create [generative summary](https://docs.vectara.com/docs/learn/grounded-generation/grounded-generation-overview), based on the retrieved documents, including citations.\n",
+    "[Vectara](https://vectara.com/) provides a Trusted Generative AI platform, allowing organizations to rapidly create a ChatGPT-like experience (an AI assistant) which is grounded in the data, documents, and knowledge that they have (technically, it is Retrieval-Augmented-Generation-as-a-service). \n",
+    "\n",
+    "Vectara serverless RAG-as-a-service provides all the components of RAG behind an easy-to-use API, including:\n",
+    "1. A way to extract text from files (PDF, PPT, DOCX, etc)\n",
+    "2. ML-based chunking that provides state of the art performance.\n",
+    "3. The [Boomerang](https://vectara.com/how-boomerang-takes-retrieval-augmented-generation-to-the-next-level-via-grounded-generation/) embeddings model.\n",
+    "4. Its own internal vector database where text chunks and embedding vectors are stored.\n",
+    "5. A query service that automatically encodes the query into embedding, and retrieves the most relevant text segments (including support for [Hybrid Search](https://docs.vectara.com/docs/api-reference/search-apis/lexical-matching) and [MMR](https://vectara.com/get-diverse-results-and-comprehensive-summaries-with-vectaras-mmr-reranker/))\n",
+    "7. An LLM to for creating a [generative summary](https://docs.vectara.com/docs/learn/grounded-generation/grounded-generation-overview), based on the retrieved documents (context), including citations.\n",
    "\n",
    "See the [Vectara API documentation](https://docs.vectara.com/docs/) for more information on how to use the API.\n",
    "\n",
@@ -25,19 +27,19 @@
   "id": "68e75fb9",
   "metadata": {},
   "source": [
-    "# Setup\n",
+    "# Getting Started\n",
    "\n",
-    "You will need a `Vectara` account to use `Vectara` with `LangChain`. To get started, use the following steps (see our [quickstart](https://docs.vectara.com/docs/quickstart) guide):\n",
-    "1. [Sign up](https://console.vectara.com/signup) for a `Vectara` account if you don't already have one. Once you have completed your sign up you will have a Vectara customer ID. You can find your customer ID by clicking on your name, on the top-right of the Vectara console window.\n",
-    "2. Within your account you can create one or more corpora. Each corpus represents an area that stores text data upon ingesting from input documents. To create a corpus, use the **\"Create Corpus\"** button. You then provide a name to your corpus as well as a description. Optionally you can define filtering attributes and apply some advanced options. If you click on your created corpus, you can see its name and corpus ID right on the top.\n",
-    "3. Next you'll need to create API keys to access the corpus. Click on the **\"Authorization\"** tab in the corpus view and then the **\"Create API Key\"** button. Give your key a name, and choose whether you want query only or query+index for your key. Click \"Create\" and you now have an active API key. Keep this key confidential. \n",
+    "To get started, use the following steps:\n",
+    "1. If you don't already have one, [Sign up](https://www.vectara.com/integrations/langchain) for your free Vectara account. Once you have completed your sign up you will have a Vectara customer ID. You can find your customer ID by clicking on your name, on the top-right of the Vectara console window.\n",
+    "2. Within your account you can create one or more corpora. Each corpus represents an area that stores text data upon ingest from input documents. To create a corpus, use the **\"Create Corpus\"** button. You then provide a name to your corpus as well as a description. Optionally you can define filtering attributes and apply some advanced options. If you click on your created corpus, you can see its name and corpus ID right on the top.\n",
+    "3. Next you'll need to create API keys to access the corpus. Click on the **\"Access Control\"** tab in the corpus view and then the **\"Create API Key\"** button. Give your key a name, and choose whether you want query-only or query+index for your key. Click \"Create\" and you now have an active API key. Keep this key confidential. \n",
    "\n",
-    "To use LangChain with Vectara, you need three values: customer ID, corpus ID and api_key.\n",
+    "To use LangChain with Vectara, you'll need to have these three values: `customer ID`, `corpus ID` and `api_key`.\n",
    "You can provide those to LangChain in two ways:\n",
    "\n",
    "1. Include in your environment these three variables: `VECTARA_CUSTOMER_ID`, `VECTARA_CORPUS_ID` and `VECTARA_API_KEY`.\n",
    "\n",
-    "> For example, you can set these variables using `os.environ` and `getpass` as follows:\n",
+    "   For example, you can set these variables using os.environ and getpass as follows:\n",
    "\n",
    "```python\n",
    "import os\n",
@@ -48,17 +50,18 @@
    "os.environ[\"VECTARA_API_KEY\"] = getpass.getpass(\"Vectara API Key:\")\n",
    "```\n",
    "\n",
-    "1. Provide them as arguments when creating the `Vectara` vectorstore object:\n",
+    "2. Add them to the `Vectara` vectorstore constructor:\n",
    "\n",
    "```python\n",
-    "vectorstore = Vectara(\n",
+    "vectara = Vectara(\n",
    "                vectara_customer_id=vectara_customer_id,\n",
    "                vectara_corpus_id=vectara_corpus_id,\n",
    "                vectara_api_key=vectara_api_key\n",
    "            )\n",
    "```\n",
+    "In this notebook we assume they are provided in the environment.\n",
    "\n",
-    "**Note:** The self-query retriever requires you to have `lark` installed (`pip install lark`). "
+    "**Notes:** The self-query retriever requires you to have `lark` installed (`pip install lark`). "
   ]
  },
  {
@@ -68,34 +71,44 @@
   "source": [
    "## Connecting to Vectara from LangChain\n",
    "\n",
-    "In this example, we assume that you've created an account and a corpus, and added your VECTARA_CUSTOMER_ID, VECTARA_CORPUS_ID and VECTARA_API_KEY (created with permissions for both indexing and query) as environment variables.\n",
+    "In this example, we assume that you've created an account and a corpus, and added your `VECTARA_CUSTOMER_ID`, `VECTARA_CORPUS_ID` and `VECTARA_API_KEY` (created with permissions for both indexing and query) as environment variables.\n",
    "\n",
-    "The corpus has 4 fields defined as metadata for filtering: year, director, rating, and genre\n"
+    "We further assume the corpus has 4 fields defined as filterable metadata attributes: `year`, `director`, `rating`, and `genre`"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 1,
+   "id": "9d3aa44f",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "import os\n",
+    "\n",
+    "os.environ[\"VECTARA_API_KEY\"] = \"<YOUR_VECTARA_API_KEY>\"\n",
+    "os.environ[\"VECTARA_CORPUS_ID\"] = \"<YOUR_VECTARA_CORPUS_ID>\"\n",
+    "os.environ[\"VECTARA_CUSTOMER_ID\"] = \"<YOUR_VECTARA_CUSTOMER_ID>\"\n",
+    "\n",
+    "from langchain.chains.query_constructor.base import AttributeInfo\n",
+    "from langchain.retrievers.self_query.base import SelfQueryRetriever\n",
+    "from langchain.schema import Document\n",
+    "from langchain_community.vectorstores import Vectara\n",
+    "from langchain_openai.chat_models import ChatOpenAI"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "id": "13a6be33-de3c-4628-acc8-b94102c275b7",
+   "metadata": {},
+   "source": [
+    "## Dataset\n",
+    "\n",
+    "We first define an example dataset of movie, and upload those to the corpus, along with the metadata:"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 2,
-   "id": "cb4a5787",
-   "metadata": {
-    "tags": []
-   },
-   "outputs": [],
-   "source": [
-    "from langchain.chains import ConversationalRetrievalChain\n",
-    "from langchain.chains.query_constructor.base import AttributeInfo\n",
-    "from langchain.retrievers.self_query.base import SelfQueryRetriever\n",
-    "from langchain_community.document_loaders import TextLoader\n",
-    "from langchain_community.embeddings import FakeEmbeddings\n",
-    "from langchain_community.vectorstores import Vectara\n",
-    "from langchain_core.documents import Document\n",
-    "from langchain_openai import OpenAI\n",
-    "from langchain_text_splitters import CharacterTextSplitter"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 3,
   "id": "bcbe04d9",
   "metadata": {
    "tags": []
@@ -136,11 +149,7 @@
    "\n",
    "vectara = Vectara()\n",
    "for doc in docs:\n",
-    "    vectara.add_texts(\n",
-    "        [doc.page_content],\n",
-    "        embedding=FakeEmbeddings(size=768),\n",
-    "        doc_metadata=doc.metadata,\n",
-    "    )"
+    "    vectara.add_texts([doc.page_content], doc_metadata=doc.metadata)"
   ]
  },
  {
@@ -148,23 +157,21 @@
   "id": "5ecaab6d",
   "metadata": {},
   "source": [
-    "## Creating our self-querying retriever\n",
-    "Now we can instantiate our retriever. To do this we'll need to provide some information upfront about the metadata fields that our documents support and a short description of the document contents."
+    "## Creating the self-querying retriever\n",
+    "Now we can instantiate our retriever. To do this we'll need to provide some information upfront about the metadata fields that our documents support and a short description of the document contents.\n",
+    "\n",
+    "We then provide an llm (in this case OpenAI) and the `vectara` vectorstore as arguments:"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 4,
+   "execution_count": 3,
   "id": "86e34dbf",
   "metadata": {
    "tags": []
   },
   "outputs": [],
   "source": [
-    "from langchain.chains.query_constructor.base import AttributeInfo\n",
-    "from langchain.retrievers.self_query.base import SelfQueryRetriever\n",
-    "from langchain_openai import OpenAI\n",
-    "\n",
    "metadata_field_info = [\n",
    "    AttributeInfo(\n",
    "        name=\"genre\",\n",
@@ -186,7 +193,7 @@
    "    ),\n",
    "]\n",
    "document_content_description = \"Brief summary of a movie\"\n",
-    "llm = OpenAI(temperature=0)\n",
+    "llm = ChatOpenAI(temperature=0, model=\"gpt-4o\", max_tokens=4069)\n",
    "retriever = SelfQueryRetriever.from_llm(\n",
    "    llm, vectara, document_content_description, metadata_field_info, verbose=True\n",
    ")"
@@ -197,13 +204,13 @@
   "id": "ea9df8d4",
   "metadata": {},
   "source": [
-    "## Testing it out\n",
+    "## Self-retrieval Queries\n",
    "And now we can try actually using our retriever!"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 5,
+   "execution_count": 4,
   "id": "38a126e9",
   "metadata": {},
   "outputs": [
@@ -211,26 +218,26 @@
     "data": {
      "text/plain": [
       "[Document(page_content='A bunch of scientists bring back dinosaurs and mayhem breaks loose', metadata={'lang': 'eng', 'offset': '0', 'len': '66', 'year': '1993', 'rating': '7.7', 'genre': 'science fiction', 'source': 'langchain'}),\n",
-       " Document(page_content='Toys come alive and have a blast doing so', metadata={'lang': 'eng', 'offset': '0', 'len': '41', 'year': '1995', 'genre': 'animated', 'source': 'langchain'}),\n",
       " Document(page_content='A psychologist / detective gets lost in a series of dreams within dreams within dreams and Inception reused the idea', metadata={'lang': 'eng', 'offset': '0', 'len': '116', 'year': '2006', 'director': 'Satoshi Kon', 'rating': '8.6', 'source': 'langchain'}),\n",
-       " Document(page_content='Leo DiCaprio gets lost in a dream within a dream within a dream within a ...', metadata={'lang': 'eng', 'offset': '0', 'len': '76', 'year': '2010', 'director': 'Christopher Nolan', 'rating': '8.2', 'source': 'langchain'}),\n",
+       " Document(page_content='Toys come alive and have a blast doing so', metadata={'lang': 'eng', 'offset': '0', 'len': '41', 'year': '1995', 'genre': 'animated', 'source': 'langchain'}),\n",
+       " Document(page_content='Three men walk into the Zone, three men walk out of the Zone', metadata={'lang': 'eng', 'offset': '0', 'len': '60', 'year': '1979', 'rating': '9.9', 'director': 'Andrei Tarkovsky', 'genre': 'science fiction', 'source': 'langchain'}),\n",
       " Document(page_content='A bunch of normal-sized women are supremely wholesome and some men pine after them', metadata={'lang': 'eng', 'offset': '0', 'len': '82', 'year': '2019', 'director': 'Greta Gerwig', 'rating': '8.3', 'source': 'langchain'}),\n",
-       " Document(page_content='Three men walk into the Zone, three men walk out of the Zone', metadata={'lang': 'eng', 'offset': '0', 'len': '60', 'year': '1979', 'rating': '9.9', 'director': 'Andrei Tarkovsky', 'genre': 'science fiction', 'source': 'langchain'})]"
+       " Document(page_content='Leo DiCaprio gets lost in a dream within a dream within a dream within a ...', metadata={'lang': 'eng', 'offset': '0', 'len': '76', 'year': '2010', 'director': 'Christopher Nolan', 'rating': '8.2', 'source': 'langchain'})]"
      ]
     },
-     "execution_count": 5,
+     "execution_count": 4,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
    "# This example only specifies a relevant query\n",
-    "retriever.invoke(\"What are some movies about dinosaurs\")"
+    "retriever.invoke(\"What are movies about scientists\")"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 6,
+   "execution_count": 5,
   "id": "fc3f1e6e",
   "metadata": {},
   "outputs": [
@@ -241,7 +248,7 @@
       " Document(page_content='Three men walk into the Zone, three men walk out of the Zone', metadata={'lang': 'eng', 'offset': '0', 'len': '60', 'year': '1979', 'rating': '9.9', 'director': 'Andrei Tarkovsky', 'genre': 'science fiction', 'source': 'langchain'})]"
      ]
     },
-     "execution_count": 6,
+     "execution_count": 5,
     "metadata": {},
     "output_type": "execute_result"
    }
@@ -253,7 +260,7 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 7,
+   "execution_count": 6,
   "id": "b19d4da0",
   "metadata": {},
   "outputs": [
@@ -263,7 +270,7 @@
       "[Document(page_content='A bunch of normal-sized women are supremely wholesome and some men pine after them', metadata={'lang': 'eng', 'offset': '0', 'len': '82', 'year': '2019', 'director': 'Greta Gerwig', 'rating': '8.3', 'source': 'langchain'})]"
      ]
     },
-     "execution_count": 7,
+     "execution_count": 6,
     "metadata": {},
     "output_type": "execute_result"
    }
@@ -275,17 +282,18 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 8,
+   "execution_count": 7,
   "id": "f900e40e",
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
-       "[Document(page_content='Three men walk into the Zone, three men walk out of the Zone', metadata={'lang': 'eng', 'offset': '0', 'len': '60', 'year': '1979', 'rating': '9.9', 'director': 'Andrei Tarkovsky', 'genre': 'science fiction', 'source': 'langchain'})]"
+       "[Document(page_content='A psychologist / detective gets lost in a series of dreams within dreams within dreams and Inception reused the idea', metadata={'lang': 'eng', 'offset': '0', 'len': '116', 'year': '2006', 'director': 'Satoshi Kon', 'rating': '8.6', 'source': 'langchain'}),\n",
+       " Document(page_content='Three men walk into the Zone, three men walk out of the Zone', metadata={'lang': 'eng', 'offset': '0', 'len': '60', 'year': '1979', 'rating': '9.9', 'director': 'Andrei Tarkovsky', 'genre': 'science fiction', 'source': 'langchain'})]"
      ]
     },
-     "execution_count": 8,
+     "execution_count": 7,
     "metadata": {},
     "output_type": "execute_result"
    }
@@ -297,17 +305,18 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 9,
+   "execution_count": 8,
   "id": "12a51522",
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
-       "[Document(page_content='Toys come alive and have a blast doing so', metadata={'lang': 'eng', 'offset': '0', 'len': '41', 'year': '1995', 'genre': 'animated', 'source': 'langchain'})]"
+       "[Document(page_content='Toys come alive and have a blast doing so', metadata={'lang': 'eng', 'offset': '0', 'len': '41', 'year': '1995', 'genre': 'animated', 'source': 'langchain'}),\n",
+       " Document(page_content='A bunch of scientists bring back dinosaurs and mayhem breaks loose', metadata={'lang': 'eng', 'offset': '0', 'len': '66', 'year': '1993', 'rating': '7.7', 'genre': 'science fiction', 'source': 'langchain'})]"
      ]
     },
-     "execution_count": 9,
+     "execution_count": 8,
     "metadata": {},
     "output_type": "execute_result"
    }
@@ -333,7 +342,7 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 10,
+   "execution_count": 9,
   "id": "bff36b88-b506-4877-9c63-e5a1a8d78e64",
   "metadata": {
    "tags": []
@@ -350,9 +359,17 @@
    ")"
   ]
  },
+  {
+   "cell_type": "markdown",
+   "id": "00e8baad-a9d7-4498-bd8d-ca41d0691386",
+   "metadata": {},
+   "source": [
+    "This is cool, we can include the number of results we would like to see in the query and the self retriever would correctly understand it. For example, let's look for "
+   ]
+  },
  {
   "cell_type": "code",
-   "execution_count": 11,
+   "execution_count": 10,
   "id": "2758d229-4f97-499c-819f-888acaf8ee10",
   "metadata": {
    "tags": []
@@ -361,19 +378,27 @@
    {
     "data": {
      "text/plain": [
-       "[Document(page_content='A bunch of scientists bring back dinosaurs and mayhem breaks loose', metadata={'lang': 'eng', 'offset': '0', 'len': '66', 'year': '1993', 'rating': '7.7', 'genre': 'science fiction', 'source': 'langchain'}),\n",
-       " Document(page_content='Toys come alive and have a blast doing so', metadata={'lang': 'eng', 'offset': '0', 'len': '41', 'year': '1995', 'genre': 'animated', 'source': 'langchain'})]"
+       "[Document(page_content='A psychologist / detective gets lost in a series of dreams within dreams within dreams and Inception reused the idea', metadata={'lang': 'eng', 'offset': '0', 'len': '116', 'year': '2006', 'director': 'Satoshi Kon', 'rating': '8.6', 'source': 'langchain'}),\n",
+       " Document(page_content='Three men walk into the Zone, three men walk out of the Zone', metadata={'lang': 'eng', 'offset': '0', 'len': '60', 'year': '1979', 'rating': '9.9', 'director': 'Andrei Tarkovsky', 'genre': 'science fiction', 'source': 'langchain'})]"
      ]
     },
-     "execution_count": 11,
+     "execution_count": 10,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
    "# This example only specifies a relevant query\n",
-    "retriever.invoke(\"what are two movies about dinosaurs\")"
+    "retriever.invoke(\"what are two movies with a rating above 8.5\")"
   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "ed4b9dbc-e3cd-442d-b108-705295f51fa1",
+   "metadata": {},
+   "outputs": [],
+   "source": []
  }
 ],
 "metadata": {
@@ -392,7 +417,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
-   "version": "3.10.12"
+   "version": "3.11.8"
  }
 },
 "nbformat": 4,
--- a/docs/docs/integrations/stores/elasticsearch.ipynb
+++ b/docs/docs/integrations/stores/elasticsearch.ipynb
@@ -0,0 +1,139 @@
+{
+ "cells": [
+  {
+   "cell_type": "raw",
+   "metadata": {},
+   "source": [
+    "---\n",
+    "sidebar_label: Elasticsearch \n",
+    "---"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "# ElasticsearchEmbeddingsCache\n",
+    "\n",
+    "The `ElasticsearchEmbeddingsCache` is a `ByteStore` implementation that uses your Elasticsearch instance for efficient storage and retrieval of embeddings.\n",
+    "\n",
+    "\n",
+    "First install the LangChain integration with Elasticsearch."
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "%pip install -U langchain-elasticsearch"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": "it can be instantiated using `CacheBackedEmbeddings.from_bytes_store` method."
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "from langchain.embeddings import CacheBackedEmbeddings\n",
+    "from langchain_elasticsearch import ElasticsearchEmbeddingsCache\n",
+    "from langchain_openai import OpenAIEmbeddings\n",
+    "\n",
+    "underlying_embeddings = OpenAIEmbeddings(model=\"text-embedding-3-small\")\n",
+    "\n",
+    "store = ElasticsearchEmbeddingsCache(\n",
+    "    es_url=\"http://localhost:9200\",\n",
+    "    index_name=\"llm-chat-cache\",\n",
+    "    metadata={\"project\": \"my_chatgpt_project\"},\n",
+    "    namespace=\"my_chatgpt_project\",\n",
+    ")\n",
+    "\n",
+    "embeddings = CacheBackedEmbeddings.from_bytes_store(\n",
+    "    underlying_embeddings=OpenAIEmbeddings(),\n",
+    "    document_embedding_cache=store,\n",
+    "    query_embedding_cache=store,\n",
+    ")"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "The index_name parameter can also accept aliases. This allows to use the ILM: Manage the index lifecycle that we suggest to consider for managing retention and controlling cache growth.\n",
+    "\n",
+    "Look at the class docstring for all parameters."
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "## Index the generated vectors\n",
+    "The cached vectors won't be searchable by default. The developer can customize the building of the Elasticsearch document in order to add indexed vector field.\n",
+    "\n",
+    "This can be done by subclassing end overriding methods. "
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "from typing import Any, Dict, List\n",
+    "\n",
+    "from langchain_elasticsearch import ElasticsearchEmbeddingsCache\n",
+    "\n",
+    "\n",
+    "class SearchableElasticsearchStore(ElasticsearchEmbeddingsCache):\n",
+    "    @property\n",
+    "    def mapping(self) -> Dict[str, Any]:\n",
+    "        mapping = super().mapping\n",
+    "        mapping[\"mappings\"][\"properties\"][\"vector\"] = {\n",
+    "            \"type\": \"dense_vector\",\n",
+    "            \"dims\": 1536,\n",
+    "            \"index\": True,\n",
+    "            \"similarity\": \"dot_product\",\n",
+    "        }\n",
+    "        return mapping\n",
+    "\n",
+    "    def build_document(self, llm_input: str, vector: List[float]) -> Dict[str, Any]:\n",
+    "        body = super().build_document(llm_input, vector)\n",
+    "        body[\"vector\"] = vector\n",
+    "        return body"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": "When overriding the mapping and the document building, please only make additive modifications, keeping the base mapping intact."
+  }
+ ],
+ "metadata": {
+  "kernelspec": {
+   "display_name": ".venv",
+   "language": "python",
+   "name": "python3"
+  },
+  "language_info": {
+   "codemirror_mode": {
+    "name": "ipython",
+    "version": 3
+   },
+   "file_extension": ".py",
+   "mimetype": "text/x-python",
+   "name": "python",
+   "nbconvert_exporter": "python",
+   "pygments_lexer": "ipython3",
+   "version": "3.11.4"
+  }
+ },
+ "nbformat": 4,
+ "nbformat_minor": 2
+}
--- a/docs/docs/integrations/text_embedding/bge_huggingface.ipynb
+++ b/docs/docs/integrations/text_embedding/bge_huggingface.ipynb
@@ -42,6 +42,12 @@
    ")"
   ]
  },
+  {
+   "metadata": {},
+   "cell_type": "markdown",
+   "source": "Note that you need to pass `query_instruction=\"\"` for `model_name=\"BAAI/bge-m3\"` see [FAQ BGE M3](https://huggingface.co/BAAI/bge-m3#faq). ",
+   "id": "f35d54e529c4cb77"
+  },
  {
   "cell_type": "code",
   "execution_count": 5,
--- a/docs/docs/integrations/text_embedding/ibm_watsonx.ipynb
+++ b/docs/docs/integrations/text_embedding/ibm_watsonx.ipynb
@@ -150,7 +150,7 @@
    "    username=\"PASTE YOUR USERNAME HERE\",\n",
    "    password=\"PASTE YOUR PASSWORD HERE\",\n",
    "    instance_id=\"openshift\",\n",
-    "    version=\"5.0\",\n",
+    "    version=\"4.8\",\n",
    "    project_id=\"PASTE YOUR PROJECT_ID HERE\",\n",
    "    params=embed_params,\n",
    ")"
--- a/docs/docs/integrations/text_embedding/jina.ipynb
+++ b/docs/docs/integrations/text_embedding/jina.ipynb
@@ -10,6 +10,16 @@
    "Let's load the Jina Embedding class."
   ]
  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "2f0a1567-6273-47a3-b19d-c30af2470810",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "pip install -U langchain-community"
+   ]
+  },
  {
   "cell_type": "code",
   "execution_count": null,
@@ -17,7 +27,11 @@
   "metadata": {},
   "outputs": [],
   "source": [
-    "from langchain_community.embeddings import JinaEmbeddings"
+    "import requests\n",
+    "from langchain_community.embeddings import JinaEmbeddings\n",
+    "from numpy import dot\n",
+    "from numpy.linalg import norm\n",
+    "from PIL import Image"
   ]
  },
  {
@@ -27,9 +41,11 @@
   "metadata": {},
   "outputs": [],
   "source": [
-    "embeddings = JinaEmbeddings(\n",
+    "text_embeddings = JinaEmbeddings(\n",
    "    jina_api_key=\"jina_*\", model_name=\"jina-embeddings-v2-base-en\"\n",
-    ")"
+    ")\n",
+    "\n",
+    "image_embeddings = JinaEmbeddings(jina_api_key=\"jina_*\", model_name=\"jina-clip-v1\")"
   ]
  },
  {
@@ -39,7 +55,15 @@
   "metadata": {},
   "outputs": [],
   "source": [
-    "text = \"This is a test document.\""
+    "text = \"This is a test document.\"\n",
+    "\n",
+    "image = \"https://avatars.githubusercontent.com/u/126733545?v=4\"\n",
+    "\n",
+    "description = \"Logo of a parrot and a chain on green background\"\n",
+    "\n",
+    "im = Image.open(requests.get(image, stream=True).raw)\n",
+    "print(\"Image:\")\n",
+    "display(im)"
   ]
  },
  {
@@ -49,7 +73,7 @@
   "metadata": {},
   "outputs": [],
   "source": [
-    "query_result = embeddings.embed_query(text)"
+    "query_result = text_embeddings.embed_query(text)"
   ]
  },
  {
@@ -69,7 +93,7 @@
   "metadata": {},
   "outputs": [],
   "source": [
-    "doc_result = embeddings.embed_documents([text])"
+    "doc_result = text_embeddings.embed_documents([text])"
   ]
  },
  {
@@ -81,6 +105,76 @@
   "source": [
    "print(doc_result)"
   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "ac2aace1-27af-4c4f-96f8-8e8b20d95b98",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "image_result = image_embeddings.embed_images([image])"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "6687808c-1977-4128-a960-888bb82c46e1",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "print(image_result)"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "2844ef7c-cf9b-4e28-b627-09887aaa0a6d",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "description_result = image_embeddings.embed_documents([description])"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "23372332-2ea3-4e4a-abc8-8307d45ebc95",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "print(description_result)"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "08d3ba5e-8957-4b10-97e3-40359ab165a6",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "cosine_similarity = dot(image_result[0], description_result[0]) / (\n",
+    "    norm(image_result[0]) * norm(description_result[0])\n",
+    ")"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "6be56ff9-774b-4347-a5cf-57d8db9e2cf2",
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "print(cosine_similarity)"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "id": "7f280807-a02b-4d4e-8ebd-01be33117999",
+   "metadata": {},
+   "outputs": [],
+   "source": []
  }
 ],
 "metadata": {
@@ -99,7 +193,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
-   "version": "3.10.11"
+   "version": "3.12.2"
  }
 },
 "nbformat": 4,
--- a/docs/docs/integrations/text_embedding/ovhcloud.ipynb
+++ b/docs/docs/integrations/text_embedding/ovhcloud.ipynb
--- a/docs/docs/integrations/text_embedding/voyageai.ipynb
+++ b/docs/docs/integrations/text_embedding/voyageai.ipynb
@@ -33,7 +33,9 @@
    "- `voyage-code-2`\n",
    "- `voyage-2`\n",
    "- `voyage-law-2`\n",
-    "- `voyage-large-2-instruct`"
+    "- `voyage-large-2-instruct`\n",
+    "- `voyage-finance-2`\n",
+    "- `voyage-multilingual-2`"
   ]
  },
  {
--- a/Show More
+++ b/Show More