Add agent

Confirm tool use
Use OAI fxns for message processing
2026-02-08 02:00:06 +00:00 · 2024-06-12 21:53:05 -07:00 · 2024-06-11 11:27:17 -07:00 · 2024-06-10 07:37:03 -07:00 · 2024-06-07 15:43:17 -07:00 · 2024-06-07 15:22:06 -07:00
565 changed files with 238873 additions and 20596 deletions
--- a/.github/scripts/check_diff.py
+++ b/.github/scripts/check_diff.py
@@ -1,11 +1,7 @@
 import json
 import sys
 import os
-from typing import Dict, List, Set
-
-import tomllib
-from collections import defaultdict
-import glob
+from typing import Dict

 LANGCHAIN_DIRS = [
    "libs/core",
@@ -15,34 +11,6 @@ LANGCHAIN_DIRS = [
    "libs/experimental",
 ]

-def dependents_graph() -> dict:
-    dependents = defaultdict(set)
-
-    for path in glob.glob("./libs/**/pyproject.toml", recursive=True):
-        if "template" in path:
-            continue
-        with open(path, "rb") as f:
-            pyproject = tomllib.load(f)['tool']['poetry']
-        pkg_dir = "libs" + "/".join(path.split("libs")[1].split("/")[:-1])
-        for dep in pyproject['dependencies']:
-            if "langchain" in dep:
-                dependents[dep].add(pkg_dir)
-    return dependents
-
-
-def add_dependents(dirs_to_eval: Set[str], dependents: dict) -> List[str]:
-    updated = set()
-    for dir_ in dirs_to_eval:
-        # handle core manually because it has so many dependents
-        if "core" in dir_:
-            updated.add(dir_)
-            continue
-        pkg = "langchain-" + dir_.split("/")[-1]
-        updated.update(dependents[pkg])
-        updated.add(dir_)
-    return list(updated)
-
-
 if __name__ == "__main__":
    files = sys.argv[1:]

@@ -113,13 +81,11 @@ if __name__ == "__main__":
                docs_edited = True
            dirs_to_run["lint"].add(".")

-    dependents = dependents_graph()
-
    outputs = {
-        "dirs-to-lint": add_dependents(
-            dirs_to_run["lint"] | dirs_to_run["test"] | dirs_to_run["extended-test"], dependents
+        "dirs-to-lint": list(
+            dirs_to_run["lint"] | dirs_to_run["test"] | dirs_to_run["extended-test"]
        ),
-        "dirs-to-test": add_dependents(dirs_to_run["test"] | dirs_to_run["extended-test"], dependents),
+        "dirs-to-test": list(dirs_to_run["test"] | dirs_to_run["extended-test"]),
        "dirs-to-extended-test": list(dirs_to_run["extended-test"]),
        "docs-edited": "true" if docs_edited else "",
    }
--- a/.github/workflows/_compile_integration_test.yml
+++ b/.github/workflows/_compile_integration_test.yml
@@ -24,7 +24,6 @@ jobs:
          - "3.9"
          - "3.10"
          - "3.11"
-          - "3.12"
    name: "poetry run pytest -m compile tests/integration_tests #${{ matrix.python-version }}"
    steps:
      - uses: actions/checkout@v4
--- a/.github/workflows/_dependencies.yml
+++ b/.github/workflows/_dependencies.yml
@@ -28,7 +28,6 @@ jobs:
          - "3.9"
          - "3.10"
          - "3.11"
-          - "3.12"
    name: dependency checks ${{ matrix.python-version }}
    steps:
      - uses: actions/checkout@v4
--- a/.github/workflows/_integration_test.yml
+++ b/.github/workflows/_integration_test.yml
@@ -12,6 +12,7 @@ env:

 jobs:
  build:
+    environment: Scheduled testing
    defaults:
      run:
        working-directory: ${{ inputs.working-directory }}
@@ -52,15 +53,8 @@ jobs:
        shell: bash
        env:
          AI21_API_KEY: ${{ secrets.AI21_API_KEY }}
-          FIREWORKS_API_KEY: ${{ secrets.FIREWORKS_API_KEY }}
          GOOGLE_API_KEY: ${{ secrets.GOOGLE_API_KEY }}
          ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
-          AZURE_OPENAI_API_VERSION: ${{ secrets.AZURE_OPENAI_API_VERSION }}
-          AZURE_OPENAI_API_BASE: ${{ secrets.AZURE_OPENAI_API_BASE }}
-          AZURE_OPENAI_API_KEY: ${{ secrets.AZURE_OPENAI_API_KEY }}
-          AZURE_OPENAI_CHAT_DEPLOYMENT_NAME: ${{ secrets.AZURE_OPENAI_CHAT_DEPLOYMENT_NAME }}
-          AZURE_OPENAI_LLM_DEPLOYMENT_NAME: ${{ secrets.AZURE_OPENAI_LLM_DEPLOYMENT_NAME }}
-          AZURE_OPENAI_EMBEDDINGS_DEPLOYMENT_NAME: ${{ secrets.AZURE_OPENAI_EMBEDDINGS_DEPLOYMENT_NAME }}
          MISTRAL_API_KEY: ${{ secrets.MISTRAL_API_KEY }}
          TOGETHER_API_KEY: ${{ secrets.TOGETHER_API_KEY }}
          OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
--- a/.github/workflows/_lint.yml
+++ b/.github/workflows/_lint.yml
@@ -34,7 +34,7 @@ jobs:
        # so linting on fewer versions makes CI faster.
        python-version:
          - "3.8"
-          - "3.12"
+          - "3.11"
    steps:
      - uses: actions/checkout@v4

--- a/.github/workflows/_test.yml
+++ b/.github/workflows/_test.yml
@@ -28,7 +28,6 @@ jobs:
          - "3.9"
          - "3.10"
          - "3.11"
-          - "3.12"
    name: "make test #${{ matrix.python-version }}"
    steps:
      - uses: actions/checkout@v4
--- a/.github/workflows/_test_doc_imports.yml
+++ b/.github/workflows/_test_doc_imports.yml
@@ -12,7 +12,7 @@ jobs:
    strategy:
      matrix:
        python-version:
-          - "3.12"
+          - "3.11"
    name: "check doc imports #${{ matrix.python-version }}"
    steps:
      - uses: actions/checkout@v4
--- a/.github/workflows/check-broken-links.yml
+++ b/.github/workflows/check-broken-links.yml
@@ -7,7 +7,6 @@ on:

 jobs:
  check-links:
-    if: github.repository_owner == 'langchain-ai'
    runs-on: ubuntu-latest
    steps:
      - uses: actions/checkout@v4
--- a/.github/workflows/check_diffs.yml
+++ b/.github/workflows/check_diffs.yml
@@ -26,7 +26,7 @@ jobs:
      - uses: actions/checkout@v4
      - uses: actions/setup-python@v5
        with:
-          python-version: '3.11'
+          python-version: '3.10'
      - id: files
        uses: Ana06/get-changed-files@v2.2.0
      - id: set-matrix
@@ -104,7 +104,6 @@ jobs:
          - "3.9"
          - "3.10"
          - "3.11"
-          - "3.12"
    runs-on: ubuntu-latest
    defaults:
      run:
--- a/.github/workflows/check_new_docs.yml
+++ b/.github/workflows/check_new_docs.yml
@@ -1,31 +0,0 @@
---
-name: Integration docs lint
-
-on:
-  push:
-    branches: [master]
-  pull_request:
-
-# If another push to the same PR or branch happens while this workflow is still running,
-# cancel the earlier run in favor of the next run.
-#
-# There's no point in testing an outdated version of the code. GitHub only allows
-# a limited number of job runners to be active at the same time, so it's better to cancel
-# pointless jobs early so that more useful jobs can run sooner.
-concurrency:
-  group: ${{ github.workflow }}-${{ github.ref }}
-  cancel-in-progress: true
-
-jobs:
-  build:
-    runs-on: ubuntu-latest
-    steps:
-      - uses: actions/checkout@v4
-      - uses: actions/setup-python@v5
-        with:
-          python-version: '3.10'
-      - id: files
-        uses: Ana06/get-changed-files@v2.2.0
-      - name: Check new docs
-        run: |
-          python docs/scripts/check_templates.py ${{ steps.files.outputs.added }}
--- a/.github/workflows/scheduled_test.yml
+++ b/.github/workflows/scheduled_test.yml
@@ -10,7 +10,6 @@ env:

 jobs:
  build:
-    if: github.repository_owner == 'langchain-ai'
    name: Python ${{ matrix.python-version }} - ${{ matrix.working-directory }}
    runs-on: ubuntu-latest
    strategy:
@@ -31,6 +30,7 @@ jobs:
          - "libs/partners/google-vertexai"
          - "libs/partners/google-genai"
          - "libs/partners/aws"
+          - "libs/partners/nvidia-ai-endpoints"

    steps:
      - uses: actions/checkout@v4
@@ -40,6 +40,10 @@ jobs:
        with:
          repository: langchain-ai/langchain-google
          path: langchain-google
+      - uses: actions/checkout@v4
+        with:
+          repository: langchain-ai/langchain-nvidia
+          path: langchain-nvidia
      - uses: actions/checkout@v4
        with:
          repository: langchain-ai/langchain-cohere
@@ -54,9 +58,11 @@ jobs:
          rm -rf \
            langchain/libs/partners/google-genai \
            langchain/libs/partners/google-vertexai \
+            langchain/libs/partners/nvidia-ai-endpoints \
            langchain/libs/partners/cohere
          mv langchain-google/libs/genai langchain/libs/partners/google-genai
          mv langchain-google/libs/vertexai langchain/libs/partners/google-vertexai
+          mv langchain-nvidia/libs/ai-endpoints langchain/libs/partners/nvidia-ai-endpoints
          mv langchain-cohere/libs/cohere langchain/libs/partners/cohere
          mv langchain-aws/libs/aws langchain/libs/partners/aws

@@ -116,6 +122,7 @@ jobs:
          rm -rf \
            langchain/libs/partners/google-genai \
            langchain/libs/partners/google-vertexai \
+            langchain/libs/partners/nvidia-ai-endpoints \
            langchain/libs/partners/cohere \
            langchain/libs/partners/aws

--- a/cookbook/azure_container_apps_dynamic_sessions_data_analyst.ipynb
+++ b/cookbook/azure_container_apps_dynamic_sessions_data_analyst.ipynb
@@ -273,7 +273,7 @@
   "source": [
    "# Tool schema for querying SQL db\n",
    "class create_df_from_sql(BaseModel):\n",
-    "    \"\"\"Execute a PostgreSQL SELECT statement and use the results to create a DataFrame with the given column names.\"\"\"\n",
+    "    \"\"\"Execute a PostgreSQL SELECT statement and use the results to create a DataFrame with the given colum names.\"\"\"\n",
    "\n",
    "    select_query: str = Field(..., description=\"A PostgreSQL SELECT statement.\")\n",
    "    # We're going to convert the results to a Pandas DataFrame that we pass\n",
--- a/docs/Makefile
+++ b/docs/Makefile
@@ -38,8 +38,6 @@ generate-files:

 	$(PYTHON) scripts/model_feat_table.py $(INTERMEDIATE_DIR)

-	$(PYTHON) scripts/document_loader_feat_table.py $(INTERMEDIATE_DIR)
-
 	$(PYTHON) scripts/copy_templates.py $(INTERMEDIATE_DIR)

 	wget -q https://raw.githubusercontent.com/langchain-ai/langserve/main/README.md -O $(INTERMEDIATE_DIR)/langserve.md
--- a/docs/docs/additional_resources/arxiv_references.mdx
+++ b/docs/docs/additional_resources/arxiv_references.mdx
@@ -24,22 +24,21 @@ Here you find [such papers](https://arxiv.org/search/?query=langchain&searchtype
 | `2305.08291v1` [Large Language Model Guided Tree-of-Thought](http://arxiv.org/abs/2305.08291v1) | Jieyi Long | 2023-05-15 | `API:` [langchain_experimental.tot](https://api.python.langchain.com/en/latest/experimental_api_reference.html#module-langchain_experimental.tot), `Cookbook:` [tree_of_thought](https://github.com/langchain-ai/langchain/blob/master/cookbook/tree_of_thought.ipynb)
 | `2305.04091v3` [Plan-and-Solve Prompting: Improving Zero-Shot Chain-of-Thought Reasoning by Large Language Models](http://arxiv.org/abs/2305.04091v3) | Lei Wang, Wanyu Xu, Yihuai Lan,  et al. | 2023-05-06 | `Cookbook:` [plan_and_execute_agent](https://github.com/langchain-ai/langchain/blob/master/cookbook/plan_and_execute_agent.ipynb)
 | `2304.08485v2` [Visual Instruction Tuning](http://arxiv.org/abs/2304.08485v2) | Haotian Liu, Chunyuan Li, Qingyang Wu,  et al. | 2023-04-17 | `Cookbook:` [Semi_structured_and_multi_modal_RAG](https://github.com/langchain-ai/langchain/blob/master/cookbook/Semi_structured_and_multi_modal_RAG.ipynb), [Semi_structured_multi_modal_RAG_LLaMA2](https://github.com/langchain-ai/langchain/blob/master/cookbook/Semi_structured_multi_modal_RAG_LLaMA2.ipynb)
-| `2304.03442v2` [Generative Agents: Interactive Simulacra of Human Behavior](http://arxiv.org/abs/2304.03442v2) | Joon Sung Park, Joseph C. O'Brien, Carrie J. Cai,  et al. | 2023-04-07 | `Cookbook:` [multiagent_bidding](https://github.com/langchain-ai/langchain/blob/master/cookbook/multiagent_bidding.ipynb), [generative_agents_interactive_simulacra_of_human_behavior](https://github.com/langchain-ai/langchain/blob/master/cookbook/generative_agents_interactive_simulacra_of_human_behavior.ipynb)
+| `2304.03442v2` [Generative Agents: Interactive Simulacra of Human Behavior](http://arxiv.org/abs/2304.03442v2) | Joon Sung Park, Joseph C. O'Brien, Carrie J. Cai,  et al. | 2023-04-07 | `Cookbook:` [generative_agents_interactive_simulacra_of_human_behavior](https://github.com/langchain-ai/langchain/blob/master/cookbook/generative_agents_interactive_simulacra_of_human_behavior.ipynb), [multiagent_bidding](https://github.com/langchain-ai/langchain/blob/master/cookbook/multiagent_bidding.ipynb)
 | `2303.17760v2` [CAMEL: Communicative Agents for "Mind" Exploration of Large Language Model Society](http://arxiv.org/abs/2303.17760v2) | Guohao Li, Hasan Abed Al Kader Hammoud, Hani Itani,  et al. | 2023-03-31 | `Cookbook:` [camel_role_playing](https://github.com/langchain-ai/langchain/blob/master/cookbook/camel_role_playing.ipynb)
 | `2303.17580v4` [HuggingGPT: Solving AI Tasks with ChatGPT and its Friends in Hugging Face](http://arxiv.org/abs/2303.17580v4) | Yongliang Shen, Kaitao Song, Xu Tan,  et al. | 2023-03-30 | `API:` [langchain_experimental.autonomous_agents](https://api.python.langchain.com/en/latest/experimental_api_reference.html#module-langchain_experimental.autonomous_agents), `Cookbook:` [hugginggpt](https://github.com/langchain-ai/langchain/blob/master/cookbook/hugginggpt.ipynb)
 | `2303.08774v6` [GPT-4 Technical Report](http://arxiv.org/abs/2303.08774v6) | OpenAI, Josh Achiam, Steven Adler,  et al. | 2023-03-15 | `Docs:` [docs/integrations/vectorstores/mongodb_atlas](https://python.langchain.com/docs/integrations/vectorstores/mongodb_atlas)
-| `2301.10226v4` [A Watermark for Large Language Models](http://arxiv.org/abs/2301.10226v4) | John Kirchenbauer, Jonas Geiping, Yuxin Wen,  et al. | 2023-01-24 | `API:` [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_community...OCIModelDeploymentTGI](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.oci_data_science_model_deployment_endpoint.OCIModelDeploymentTGI.html#langchain_community.llms.oci_data_science_model_deployment_endpoint.OCIModelDeploymentTGI), [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference)
+| `2301.10226v4` [A Watermark for Large Language Models](http://arxiv.org/abs/2301.10226v4) | John Kirchenbauer, Jonas Geiping, Yuxin Wen,  et al. | 2023-01-24 | `API:` [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_community...OCIModelDeploymentTGI](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.oci_data_science_model_deployment_endpoint.OCIModelDeploymentTGI.html#langchain_community.llms.oci_data_science_model_deployment_endpoint.OCIModelDeploymentTGI), [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint)
 | `2212.10496v1` [Precise Zero-Shot Dense Retrieval without Relevance Labels](http://arxiv.org/abs/2212.10496v1) | Luyu Gao, Xueguang Ma, Jimmy Lin,  et al. | 2022-12-20 | `API:` [langchain...HypotheticalDocumentEmbedder](https://api.python.langchain.com/en/latest/chains/langchain.chains.hyde.base.HypotheticalDocumentEmbedder.html#langchain.chains.hyde.base.HypotheticalDocumentEmbedder), `Template:` [hyde](https://python.langchain.com/docs/templates/hyde), `Cookbook:` [hypothetical_document_embeddings](https://github.com/langchain-ai/langchain/blob/master/cookbook/hypothetical_document_embeddings.ipynb)
 | `2212.07425v3` [Robust and Explainable Identification of Logical Fallacies in Natural Language Arguments](http://arxiv.org/abs/2212.07425v3) | Zhivar Sourati, Vishnu Priya Prasanna Venkatesh, Darshan Deshpande,  et al. | 2022-12-12 | `API:` [langchain_experimental.fallacy_removal](https://api.python.langchain.com/en/latest/experimental_api_reference.html#module-langchain_experimental.fallacy_removal)
 | `2211.13892v2` [Complementary Explanations for Effective In-Context Learning](http://arxiv.org/abs/2211.13892v2) | Xi Ye, Srinivasan Iyer, Asli Celikyilmaz,  et al. | 2022-11-25 | `API:` [langchain_core...MaxMarginalRelevanceExampleSelector](https://api.python.langchain.com/en/latest/example_selectors/langchain_core.example_selectors.semantic_similarity.MaxMarginalRelevanceExampleSelector.html#langchain_core.example_selectors.semantic_similarity.MaxMarginalRelevanceExampleSelector)
 | `2211.10435v2` [PAL: Program-aided Language Models](http://arxiv.org/abs/2211.10435v2) | Luyu Gao, Aman Madaan, Shuyan Zhou,  et al. | 2022-11-18 | `API:` [langchain_experimental...PALChain](https://api.python.langchain.com/en/latest/pal_chain/langchain_experimental.pal_chain.base.PALChain.html#langchain_experimental.pal_chain.base.PALChain), [langchain_experimental.pal_chain](https://api.python.langchain.com/en/latest/experimental_api_reference.html#module-langchain_experimental.pal_chain), `Cookbook:` [program_aided_language_model](https://github.com/langchain-ai/langchain/blob/master/cookbook/program_aided_language_model.ipynb)
-| `2210.03629v3` [ReAct: Synergizing Reasoning and Acting in Language Models](http://arxiv.org/abs/2210.03629v3) | Shunyu Yao, Jeffrey Zhao, Dian Yu,  et al. | 2022-10-06 | `Docs:` [docs/integrations/providers/cohere](https://python.langchain.com/docs/integrations/providers/cohere), [docs/integrations/chat/huggingface](https://python.langchain.com/docs/integrations/chat/huggingface), [docs/integrations/tools/ionic_shopping](https://python.langchain.com/docs/integrations/tools/ionic_shopping), `API:` [langchain...create_react_agent](https://api.python.langchain.com/en/latest/agents/langchain.agents.react.agent.create_react_agent.html#langchain.agents.react.agent.create_react_agent), [langchain...TrajectoryEvalChain](https://api.python.langchain.com/en/latest/evaluation/langchain.evaluation.agents.trajectory_eval_chain.TrajectoryEvalChain.html#langchain.evaluation.agents.trajectory_eval_chain.TrajectoryEvalChain)
 | `2209.10785v2` [Deep Lake: a Lakehouse for Deep Learning](http://arxiv.org/abs/2209.10785v2) | Sasun Hambardzumyan, Abhinav Tuli, Levon Ghukasyan,  et al. | 2022-09-22 | `Docs:` [docs/integrations/providers/activeloop_deeplake](https://python.langchain.com/docs/integrations/providers/activeloop_deeplake)
 | `2205.12654v1` [Bitext Mining Using Distilled Sentence Representations for Low-Resource Languages](http://arxiv.org/abs/2205.12654v1) | Kevin Heffernan, Onur Çelebi, Holger Schwenk | 2022-05-25 | `API:` [langchain_community...LaserEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.laser.LaserEmbeddings.html#langchain_community.embeddings.laser.LaserEmbeddings)
 | `2204.00498v1` [Evaluating the Text-to-SQL Capabilities of Large Language Models](http://arxiv.org/abs/2204.00498v1) | Nitarshan Rajkumar, Raymond Li, Dzmitry Bahdanau | 2022-03-15 | `API:` [langchain_community...SparkSQL](https://api.python.langchain.com/en/latest/utilities/langchain_community.utilities.spark_sql.SparkSQL.html#langchain_community.utilities.spark_sql.SparkSQL), [langchain_community...SQLDatabase](https://api.python.langchain.com/en/latest/utilities/langchain_community.utilities.sql_database.SQLDatabase.html#langchain_community.utilities.sql_database.SQLDatabase)
-| `2202.00666v5` [Locally Typical Sampling](http://arxiv.org/abs/2202.00666v5) | Clara Meister, Tiago Pimentel, Gian Wiher,  et al. | 2022-02-01 | `API:` [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference)
+| `2202.00666v5` [Locally Typical Sampling](http://arxiv.org/abs/2202.00666v5) | Clara Meister, Tiago Pimentel, Gian Wiher,  et al. | 2022-02-01 | `API:` [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint)
 | `2103.00020v1` [Learning Transferable Visual Models From Natural Language Supervision](http://arxiv.org/abs/2103.00020v1) | Alec Radford, Jong Wook Kim, Chris Hallacy,  et al. | 2021-02-26 | `API:` [langchain_experimental.open_clip](https://api.python.langchain.com/en/latest/experimental_api_reference.html#module-langchain_experimental.open_clip)
-| `1909.05858v2` [CTRL: A Conditional Transformer Language Model for Controllable Generation](http://arxiv.org/abs/1909.05858v2) | Nitish Shirish Keskar, Bryan McCann, Lav R. Varshney,  et al. | 2019-09-11 | `API:` [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference)
+| `1909.05858v2` [CTRL: A Conditional Transformer Language Model for Controllable Generation](http://arxiv.org/abs/1909.05858v2) | Nitish Shirish Keskar, Bryan McCann, Lav R. Varshney,  et al. | 2019-09-11 | `API:` [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint)
 | `1908.10084v1` [Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks](http://arxiv.org/abs/1908.10084v1) | Nils Reimers, Iryna Gurevych | 2019-08-27 | `Docs:` [docs/integrations/text_embedding/sentence_transformers](https://python.langchain.com/docs/integrations/text_embedding/sentence_transformers)

 ## Self-Discover: Large Language Models Self-Compose Reasoning Structures
@@ -419,7 +418,7 @@ publicly available.
 - **URL:** http://arxiv.org/abs/2304.03442v2
 - **LangChain:**

-   - **Cookbook:** [multiagent_bidding](https://github.com/langchain-ai/langchain/blob/master/cookbook/multiagent_bidding.ipynb), [generative_agents_interactive_simulacra_of_human_behavior](https://github.com/langchain-ai/langchain/blob/master/cookbook/generative_agents_interactive_simulacra_of_human_behavior.ipynb)
+   - **Cookbook:** [generative_agents_interactive_simulacra_of_human_behavior](https://github.com/langchain-ai/langchain/blob/master/cookbook/generative_agents_interactive_simulacra_of_human_behavior.ipynb), [multiagent_bidding](https://github.com/langchain-ai/langchain/blob/master/cookbook/multiagent_bidding.ipynb)

 **Abstract:** Believable proxies of human behavior can empower interactive applications
 ranging from immersive environments to rehearsal spaces for interpersonal
@@ -541,7 +540,7 @@ more than 1/1,000th the compute of GPT-4.
 - **URL:** http://arxiv.org/abs/2301.10226v4
 - **LangChain:**

-   - **API Reference:** [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_community...OCIModelDeploymentTGI](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.oci_data_science_model_deployment_endpoint.OCIModelDeploymentTGI.html#langchain_community.llms.oci_data_science_model_deployment_endpoint.OCIModelDeploymentTGI), [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference)
+   - **API Reference:** [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_community...OCIModelDeploymentTGI](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.oci_data_science_model_deployment_endpoint.OCIModelDeploymentTGI.html#langchain_community.llms.oci_data_science_model_deployment_endpoint.OCIModelDeploymentTGI), [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint)

 **Abstract:** Potential harms of large language models can be mitigated by watermarking
 model output, i.e., embedding signals into generated text that are invisible to
@@ -684,41 +683,6 @@ accuracy on the GSM8K benchmark of math word problems, surpassing PaLM-540B
 which uses chain-of-thought by absolute 15% top-1. Our code and data are
 publicly available at http://reasonwithpal.com/ .
                
-## ReAct: Synergizing Reasoning and Acting in Language Models
-
- **arXiv id:** 2210.03629v3
- **Title:** ReAct: Synergizing Reasoning and Acting in Language Models
- **Authors:** Shunyu Yao, Jeffrey Zhao, Dian Yu,  et al.
- **Published Date:** 2022-10-06
- **URL:** http://arxiv.org/abs/2210.03629v3
- **LangChain:**
-
-   - **Documentation:** [docs/integrations/providers/cohere](https://python.langchain.com/docs/integrations/providers/cohere), [docs/integrations/chat/huggingface](https://python.langchain.com/docs/integrations/chat/huggingface), [docs/integrations/tools/ionic_shopping](https://python.langchain.com/docs/integrations/tools/ionic_shopping)
-   - **API Reference:** [langchain...create_react_agent](https://api.python.langchain.com/en/latest/agents/langchain.agents.react.agent.create_react_agent.html#langchain.agents.react.agent.create_react_agent), [langchain...TrajectoryEvalChain](https://api.python.langchain.com/en/latest/evaluation/langchain.evaluation.agents.trajectory_eval_chain.TrajectoryEvalChain.html#langchain.evaluation.agents.trajectory_eval_chain.TrajectoryEvalChain)
-
-**Abstract:** While large language models (LLMs) have demonstrated impressive capabilities
-across tasks in language understanding and interactive decision making, their
-abilities for reasoning (e.g. chain-of-thought prompting) and acting (e.g.
-action plan generation) have primarily been studied as separate topics. In this
-paper, we explore the use of LLMs to generate both reasoning traces and
-task-specific actions in an interleaved manner, allowing for greater synergy
-between the two: reasoning traces help the model induce, track, and update
-action plans as well as handle exceptions, while actions allow it to interface
-with external sources, such as knowledge bases or environments, to gather
-additional information. We apply our approach, named ReAct, to a diverse set of
-language and decision making tasks and demonstrate its effectiveness over
-state-of-the-art baselines, as well as improved human interpretability and
-trustworthiness over methods without reasoning or acting components.
-Concretely, on question answering (HotpotQA) and fact verification (Fever),
-ReAct overcomes issues of hallucination and error propagation prevalent in
-chain-of-thought reasoning by interacting with a simple Wikipedia API, and
-generates human-like task-solving trajectories that are more interpretable than
-baselines without reasoning traces. On two interactive decision making
-benchmarks (ALFWorld and WebShop), ReAct outperforms imitation and
-reinforcement learning methods by an absolute success rate of 34% and 10%
-respectively, while being prompted with only one or two in-context examples.
-Project site with code: https://react-lm.github.io
-                
 ## Deep Lake: a Lakehouse for Deep Learning

 - **arXiv id:** 2209.10785v2
@@ -804,7 +768,7 @@ few-shot examples.
 - **URL:** http://arxiv.org/abs/2202.00666v5
 - **LangChain:**

-   - **API Reference:** [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference)
+   - **API Reference:** [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint)

 **Abstract:** Today's probabilistic language generators fall short when it comes to
 producing coherent and fluent text despite the fact that the underlying models
@@ -868,7 +832,7 @@ https://github.com/OpenAI/CLIP.
 - **URL:** http://arxiv.org/abs/1909.05858v2
 - **LangChain:**

-   - **API Reference:** [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference)
+   - **API Reference:** [langchain_community...HuggingFaceTextGenInference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference.html#langchain_community.llms.huggingface_text_gen_inference.HuggingFaceTextGenInference), [langchain_community...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_community.llms.huggingface_endpoint.HuggingFaceEndpoint), [langchain_huggingface...HuggingFaceEndpoint](https://api.python.langchain.com/en/latest/llms/langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint.html#langchain_huggingface.llms.huggingface_endpoint.HuggingFaceEndpoint)

 **Abstract:** Large-scale language models show promising text generation capabilities, but
 users cannot easily control particular aspects of the generated text. We
--- a/docs/docs/additional_resources/tutorials.mdx
+++ b/docs/docs/additional_resources/tutorials.mdx
@@ -11,7 +11,6 @@
 ### [by Prompt Engineering](https://www.youtube.com/playlist?list=PLVEEucA9MYhOu89CX8H3MBZqayTbcCTMr)
 ### [by Mayo Oshin](https://www.youtube.com/@chatwithdata/search?query=langchain)
 ### [by 1 little Coder](https://www.youtube.com/playlist?list=PLpdmBGJ6ELUK-v0MK-t4wZmVEbxM5xk6L)
-### [by BobLin (Chinese language)](https://www.youtube.com/playlist?list=PLbd7ntv6PxC3QMFQvtWfk55p-Op_syO1C)

 ## Courses

@@ -46,6 +45,7 @@
 - [Generative AI with LangChain](https://www.amazon.com/Generative-AI-LangChain-language-ChatGPT/dp/1835083463/ref=sr_1_1?crid=1GMOMH0G7GLR&keywords=generative+ai+with+langchain&qid=1703247181&sprefix=%2Caps%2C298&sr=8-1) by [Ben Auffrath](https://www.amazon.com/stores/Ben-Auffarth/author/B08JQKSZ7D?ref=ap_rdr&store_ref=ap_rdr&isDramIntegrated=true&shoppingPortalEnabled=true), ©️ 2023 Packt Publishing
 - [LangChain AI Handbook](https://www.pinecone.io/learn/langchain/) By **James Briggs** and **Francisco Ingham**
 - [LangChain Cheatsheet](https://pub.towardsai.net/langchain-cheatsheet-all-secrets-on-a-single-page-8be26b721cde) by **Ivan Reznikov**
- [Dive into Langchain (Chinese language)](https://langchain.boblin.app/)

 ---------------------
+
+
--- a/docs/docs/concepts.mdx
+++ b/docs/docs/concepts.mdx
@@ -11,7 +11,7 @@ LangChain as a framework consists of a number of packages.

 ### `langchain-core`
 This package contains base abstractions of different components and ways to compose them together.
-The interfaces for core components like LLMs, vector stores, retrievers and more are defined here.
+The interfaces for core components like LLMs, vectorstores, retrievers and more are defined here.
 No third party integrations are defined here.
 The dependencies are kept purposefully very lightweight.

@@ -30,7 +30,7 @@ All chains, agents, and retrieval strategies here are NOT specific to any one in

 This package contains third party integrations that are maintained by the LangChain community.
 Key partner packages are separated out (see below).
-This contains all integrations for various components (LLMs, vector stores, retrievers).
+This contains all integrations for various components (LLMs, vectorstores, retrievers).
 All dependencies in this package are optional to keep the package as lightweight as possible.

 ### [`langgraph`](https://langchain-ai.github.io/langgraph)
@@ -96,9 +96,9 @@ To make it as easy as possible to create custom chains, we've implemented a ["Ru
 This is a standard interface, which makes it easy to define custom chains as well as invoke them in a standard way.
 The standard interface includes:

- `stream`: stream back chunks of the response
- `invoke`: call the chain on an input
- `batch`: call the chain on a list of inputs
+- [`stream`](#stream): stream back chunks of the response
+- [`invoke`](#invoke): call the chain on an input
+- [`batch`](#batch): call the chain on a list of inputs

 These also have corresponding async methods that should be used with [asyncio](https://docs.python.org/3/library/asyncio.html) `await` syntax for concurrency:

@@ -133,14 +133,14 @@ Some components LangChain implements, some components we rely on third-party int
 <span data-heading-keywords="chat model,chat models"></span>

 Language models that use a sequence of messages as inputs and return chat messages as outputs (as opposed to using plain text).
-These are traditionally newer models (older models are generally `LLMs`, see below).
+These are traditionally newer models (older models are generally `LLMs`, see above).
 Chat models support the assignment of distinct roles to conversation messages, helping to distinguish messages from the AI, users, and instructions such as system messages.

 Although the underlying models are messages in, message out, the LangChain wrappers also allow these models to take a string as input. This means you can easily use chat models in place of LLMs.

-When a string is passed in as input, it is converted to a `HumanMessage` and then passed to the underlying model.
+When a string is passed in as input, it is converted to a HumanMessage and then passed to the underlying model.

-LangChain does not host any Chat Models, rather we rely on third party integrations.
+LangChain does not provide any ChatModels, rather we rely on third party integrations.

 We have some standardized parameters when constructing ChatModels:
 - `model`: the name of the model
@@ -155,27 +155,17 @@ Please see the [tool calling section](/docs/concepts/#functiontool-calling) for

 For specifics on how to use chat models, see the [relevant how-to guides here](/docs/how_to/#chat-models).

-#### Multimodality
-
-Some chat models are multimodal, accepting images, audio and even video as inputs. These are still less common, meaning model providers haven't standardized on the "best" way to define the API. Multimodal **outputs** are even less common. As such, we've kept our multimodal abstractions fairly light weight and plan to further solidify the multimodal APIs and interaction patterns as the field matures.
-
-In LangChain, most chat models that support multimodal inputs also accept those values in OpenAI's content blocks format. So far this is restricted to image inputs. For models like Gemini which support video and other bytes input, the APIs also support the native, model-specific representations.
-
-For specifics on how to use multimodal models, see the [relevant how-to guides here](/docs/how_to/#multimodal).
-
-For a full list of LangChain model providers with multimodal models, [check out this table](/docs/integrations/chat/#advanced-features).
-
 ### LLMs
 <span data-heading-keywords="llm,llms"></span>

 Language models that takes a string as input and returns a string.
-These are traditionally older models (newer models generally are [Chat Models](/docs/concepts/#chat-models), see below).
+These are traditionally older models (newer models generally are `ChatModels`, see below).

 Although the underlying models are string in, string out, the LangChain wrappers also allow these models to take messages as input.
-This gives them the same interface as [Chat Models](/docs/concepts/#chat-models).
+This makes them interchangeable with ChatModels.
 When messages are passed in as input, they will be formatted into a string under the hood before being passed to the underlying model.

-LangChain does not host any LLMs, rather we rely on third party integrations.
+LangChain does not provide any LLMs, rather we rely on third party integrations.

 For specifics on how to use LLMs, see the [relevant how-to guides here](/docs/how_to/#llms).

@@ -373,7 +363,7 @@ An essential component of a conversation is being able to refer to information i
 At bare minimum, a conversational system should be able to access some window of past messages directly.

 The concept of `ChatHistory` refers to a class in LangChain which can be used to wrap an arbitrary chain.
-This `ChatHistory` will keep track of inputs and outputs of the underlying chain, and append them as messages to a message database.
+This `ChatHistory` will keep track of inputs and outputs of the underlying chain, and append them as messages to a message database
 Future interactions will then load those messages and pass them into the chain as part of the input.

 ### Documents
@@ -425,14 +415,9 @@ For specifics on how to use text splitters, see the [relevant how-to guides here
 ### Embedding models
 <span data-heading-keywords="embedding,embeddings"></span>

-Embedding models create a vector representation of a piece of text. You can think of a vector as an array of numbers that captures the semantic meaning of the text.
-By representing the text in this way, you can perform mathematical operations that allow you to do things like search for other pieces of text that are most similar in meaning.
-These natural language search capabilities underpin many types of [context retrieval](/docs/concepts/#retrieval),
-where we provide an LLM with the relevant data it needs to effectively respond to a query.
+The Embeddings class is a class designed for interfacing with text embedding models. There are lots of embedding model providers (OpenAI, Cohere, Hugging Face, etc) - this class is designed to provide a standard interface for all of them.

-![](/img/embeddings.png)
-
-The `Embeddings` class is a class designed for interfacing with text embedding models. There are many different embedding model providers (OpenAI, Cohere, Hugging Face, etc) and local models, and this class is designed to provide a standard interface for all of them.
+Embeddings create a vector representation of a piece of text. This is useful because it means we can think about text in the vector space, and do things like semantic search where we look for pieces of text that are most similar in the vector space.

 The base Embeddings class in LangChain provides two methods: one for embedding documents and one for embedding a query. The former takes as input multiple texts, while the latter takes a single text. The reason for having these as two separate methods is that some embedding providers have different embedding methods for documents (to be searched over) vs queries (the search query itself).

@@ -445,9 +430,6 @@ One of the most common ways to store and search over unstructured data is to emb
 and then at query time to embed the unstructured query and retrieve the embedding vectors that are 'most similar' to the embedded query.
 A vector store takes care of storing embedded data and performing vector search for you.

-Most vector stores can also store metadata about embedded vectors and support filtering on that metadata before
-similarity search, allowing you more control over returned documents.
-
 Vector stores can be converted to the retriever interface by doing:

 ```python
@@ -463,7 +445,7 @@ For specifics on how to use vector stores, see the [relevant how-to guides here]
 A retriever is an interface that returns documents given an unstructured query.
 It is more general than a vector store.
 A retriever does not need to be able to store documents, only to return (or retrieve) them.
-Retrievers can be created from vector stores, but are also broad enough to include [Wikipedia search](/docs/integrations/retrievers/wikipedia/) and [Amazon Kendra](/docs/integrations/retrievers/amazon_kendra_retriever/).
+Retrievers can be created from vectorstores, but are also broad enough to include [Wikipedia search](/docs/integrations/retrievers/wikipedia/) and [Amazon Kendra](/docs/integrations/retrievers/amazon_kendra_retriever/).

 Retrievers accept a string query as input and return a list of Document's as output.

@@ -532,6 +514,14 @@ If you are still using AgentExecutor, do not fear: we still have a guide on [how
 It is recommended, however, that you start to transition to LangGraph.
 In order to assist in this we have put together a [transition guide on how to do so](/docs/how_to/migrate_agent).

+### Multimodal
+
+Some models are multimodal, accepting images, audio and even video as inputs. These are still less common, meaning model providers haven't standardized on the "best" way to define the API. Multimodal **outputs** are even less common. As such, we've kept our multimodal abstractions fairly light weight and plan to further solidify the multimodal APIs and interaction patterns as the field matures.
+
+In LangChain, most chat models that support multimodal inputs also accept those values in OpenAI's content blocks format. So far this is restricted to image inputs. For models like Gemini which support video and other bytes input, the APIs also support the native, model-specific representations.
+
+For specifics on how to use multimodal models, see the [relevant how-to guides here](/docs/how_to/#multimodal).
+
 ### Callbacks

 LangChain provides a callbacks system that allows you to hook into the various stages of your LLM application. This is useful for logging, monitoring, streaming, and other tasks.
@@ -606,220 +596,13 @@ For specifics on how to use callbacks, see the [relevant how-to guides here](/do

 ## Techniques

-### Streaming
-<span data-heading-keywords="stream,streaming"></span>
-
-Individual LLM calls often run for much longer than traditional resource requests.
-This compounds when you build more complex chains or agents that require multiple reasoning steps.
-
-Fortunately, LLMs generate output iteratively, which means it's possible to show sensible intermediate results
-before the final response is ready. Consuming output as soon as it becomes available has therefore become a vital part of the UX
-around building apps with LLMs to help alleviate latency issues, and LangChain aims to have first-class support for streaming.
-
-Below, we'll discuss some concepts and considerations around streaming in LangChain.
-
-#### `.stream()` and `.astream()`
-
-Most modules in LangChain include the `.stream()` method (and the equivalent `.astream()` method for [async](https://docs.python.org/3/library/asyncio.html) environments) as an ergonomic streaming interface.
-`.stream()` returns an iterator, which you can consume with a simple `for` loop. Here's an example with a chat model:
-
-```python
-from langchain_anthropic import ChatAnthropic
-
-model = ChatAnthropic(model="claude-3-sonnet-20240229")
-
-for chunk in model.stream("what color is the sky?"):
-    print(chunk.content, end="|", flush=True)
-```
-
-For models (or other components) that don't support streaming natively, this iterator would just yield a single chunk, but
-you could still use the same general pattern when calling them. Using `.stream()` will also automatically call the model in streaming mode
-without the need to provide additional config.
-
-The type of each outputted chunk depends on the type of component - for example, chat models yield [`AIMessageChunks`](https://api.python.langchain.com/en/latest/messages/langchain_core.messages.ai.AIMessageChunk.html).
-Because this method is part of [LangChain Expression Language](/docs/concepts/#langchain-expression-language-lcel),
-you can handle formatting differences from different outputs using an [output parser](/docs/concepts/#output-parsers) to transform
-each yielded chunk.
-
-You can check out [this guide](/docs/how_to/streaming/#using-stream) for more detail on how to use `.stream()`.
-
-#### `.astream_events()`
-<span data-heading-keywords="astream_events,stream_events,stream events"></span>
-
-While the `.stream()` method is intuitive, it can only return the final generated value of your chain. This is fine for single LLM calls,
-but as you build more complex chains of several LLM calls together, you may want to use the intermediate values of
-the chain alongside the final output - for example, returning sources alongside the final generation when building a chat
-over documents app.
-
-There are ways to do this [using callbacks](/docs/concepts/#callbacks-1), or by constructing your chain in such a way that it passes intermediate
-values to the end with something like chained [`.assign()`](/docs/how_to/passthrough/) calls, but LangChain also includes an
-`.astream_events()` method that combines the flexibility of callbacks with the ergonomics of `.stream()`. When called, it returns an iterator
-which yields [various types of events](/docs/how_to/streaming/#event-reference) that you can filter and process according
-to the needs of your project.
-
-Here's one small example that prints just events containing streamed chat model output:
-
-```python
-from langchain_core.output_parsers import StrOutputParser
-from langchain_core.prompts import ChatPromptTemplate
-from langchain_anthropic import ChatAnthropic
-
-model = ChatAnthropic(model="claude-3-sonnet-20240229")
-
-prompt = ChatPromptTemplate.from_template("tell me a joke about {topic}")
-parser = StrOutputParser()
-chain = prompt | model | parser
-
-async for event in chain.astream_events({"topic": "parrot"}, version="v2"):
-    kind = event["event"]
-    if kind == "on_chat_model_stream":
-        print(event, end="|", flush=True)
-```
-
-You can roughly think of it as an iterator over callback events (though the format differs) - and you can use it on almost all LangChain components!
-
-See [this guide](/docs/how_to/streaming/#using-stream-events) for more detailed information on how to use `.astream_events()`,
-including a table listing available events.
-
-#### Callbacks
-
-The lowest level way to stream outputs from LLMs in LangChain is via the [callbacks](/docs/concepts/#callbacks) system. You can pass a
-callback handler that handles the [`on_llm_new_token`](https://api.python.langchain.com/en/latest/callbacks/langchain.callbacks.streaming_aiter.AsyncIteratorCallbackHandler.html#langchain.callbacks.streaming_aiter.AsyncIteratorCallbackHandler.on_llm_new_token) event into LangChain components. When that component is invoked, any
-[LLM](/docs/concepts/#llms) or [chat model](/docs/concepts/#chat-models) contained in the component calls
-the callback with the generated token. Within the callback, you could pipe the tokens into some other destination, e.g. a HTTP response.
-You can also handle the [`on_llm_end`](https://api.python.langchain.com/en/latest/callbacks/langchain.callbacks.streaming_aiter.AsyncIteratorCallbackHandler.html#langchain.callbacks.streaming_aiter.AsyncIteratorCallbackHandler.on_llm_end) event to perform any necessary cleanup.
-
-You can see [this how-to section](/docs/how_to/#callbacks) for more specifics on using callbacks.
-
-Callbacks were the first technique for streaming introduced in LangChain. While powerful and generalizable,
-they can be unwieldy for developers. For example:
-
- You need to explicitly initialize and manage some aggregator or other stream to collect results.
- The execution order isn't explicitly guaranteed, and you could theoretically have a callback run after the `.invoke()` method finishes.
- Providers would often make you pass an additional parameter to stream outputs instead of returning them all at once.
- You would often ignore the result of the actual model call in favor of callback results.
-
-#### Tokens
-
-The unit that most model providers use to measure input and output is via a unit called a **token**.
-Tokens are the basic units that language models read and generate when processing or producing text.
-The exact definition of a token can vary depending on the specific way the model was trained -
-for instance, in English, a token could be a single word like "apple", or a part of a word like "app".
-
-When you send a model a prompt, the words and characters in the prompt are encoded into tokens using a **tokenizer**.
-The model then streams back generated output tokens, which the tokenizer decodes into human-readable text.
-The below example shows how OpenAI models tokenize `LangChain is cool!`:
-
-![](/img/tokenization.png)
-
-You can see that it gets split into 5 different tokens, and that the boundaries between tokens are not exactly the same as word boundaries.
-
-The reason language models use tokens rather than something more immediately intuitive like "characters"
-has to do with how they process and understand text. At a high-level, language models iteratively predict their next generated output based on
-the initial input and their previous generations. Training the model using tokens language models to handle linguistic
-units (like words or subwords) that carry meaning, rather than individual characters, which makes it easier for the model
-to learn and understand the structure of the language, including grammar and context.
-Furthermore, using tokens can also improve efficiency, since the model processes fewer units of text compared to character-level processing.
-
-### Structured output
-
-LLMs are capable of generating arbitrary text. This enables the model to respond appropriately to a wide
-range of inputs, but for some use-cases, it can be useful to constrain the LLM's output
-to a specific format or structure. This is referred to as **structured output**.
-
-For example, if the output is to be stored in a relational database,
-it is much easier if the model generates output that adheres to a defined schema or format.
-[Extracting specific information](/docs/tutorials/extraction/) from unstructured text is another
-case where this is particularly useful. Most commonly, the output format will be JSON,
-though other formats such as [YAML](/docs/how_to/output_parser_yaml/) can be useful too. Below, we'll discuss
-a few ways to get structured output from models in LangChain.
-
-#### `.with_structured_output()`
-
-For convenience, some LangChain chat models support a `.with_structured_output()` method.
-This method only requires a schema as input, and returns a dict or Pydantic object.
-Generally, this method is only present on models that support one of the more advanced methods described below,
-and will use one of them under the hood. It takes care of importing a suitable output parser and
-formatting the schema in the right format for the model.
-
-For more information, check out this [how-to guide](/docs/how_to/structured_output/#the-with_structured_output-method).
-
-#### Raw prompting
-
-The most intuitive way to get a model to structure output is to ask nicely.
-In addition to your query, you can give instructions describing what kind of output you'd like, then
-parse the output using an [output parser](/docs/concepts/#output-parsers) to convert the raw
-model message or string output into something more easily manipulated.
-
-The biggest benefit to raw prompting is its flexibility:
-
- Raw prompting does not require any special model features, only sufficient reasoning capability to understand
-the passed schema.
- You can prompt for any format you'd like, not just JSON. This can be useful if the model you
-are using is more heavily trained on a certain type of data, such as XML or YAML.
-
-However, there are some drawbacks too:
-
- LLMs are non-deterministic, and prompting a LLM to consistently output data in the exactly correct format
-for smooth parsing can be surprisingly difficult and model-specific.
- Individual models have quirks depending on the data they were trained on, and optimizing prompts can be quite difficult.
-Some may be better at interpreting [JSON schema](https://json-schema.org/), others may be best with TypeScript definitions,
-and still others may prefer XML.
-
-While we'll next go over some ways that you can take advantage of features offered by
-model providers to increase reliability, prompting techniques remain important for tuning your
-results no matter what method you choose.
-
-#### JSON mode
-<span data-heading-keywords="json mode"></span>
-
-Some models, such as [Mistral](/docs/integrations/chat/mistralai/), [OpenAI](/docs/integrations/chat/openai/),
-[Together AI](/docs/integrations/chat/together/) and [Ollama](/docs/integrations/chat/ollama/),
-support a feature called **JSON mode**, usually enabled via config.
-
-When enabled, JSON mode will constrain the model's output to always be some sort of valid JSON.
-Often they require some custom prompting, but it's usually much less burdensome and along the lines of,
-`"you must always return JSON"`, and the [output is easier to parse](/docs/how_to/output_parser_json/).
-
-It's also generally simpler and more commonly available than tool calling.
-
-Here's an example:
-
-```python
-from langchain_core.prompts import ChatPromptTemplate
-from langchain_openai import ChatOpenAI
-from langchain.output_parsers.json import SimpleJsonOutputParser
-
-model = ChatOpenAI(
-    model="gpt-4o",
-    model_kwargs={ "response_format": { "type": "json_object" } },
-)
-
-prompt = ChatPromptTemplate.from_template(
-    "Answer the user's question to the best of your ability."
-    'You must always output a JSON object with an "answer" key and a "followup_question" key.'
-    "{question}"
-)
-
-chain = prompt | model | SimpleJsonOutputParser()
-
-chain.invoke({ "question": "What is the powerhouse of the cell?" })
-```
-
-```
-{'answer': 'The powerhouse of the cell is the mitochondrion. It is responsible for producing energy in the form of ATP through cellular respiration.',
- 'followup_question': 'Would you like to know more about how mitochondria produce energy?'}
-```
-
-For a full list of model providers that support JSON mode, see [this table](/docs/integrations/chat/#advanced-features).
-
-#### Function/tool calling
+### Function/tool calling

 :::info
 We use the term tool calling interchangeably with function calling. Although
 function calling is sometimes meant to refer to invocations of a single function,
 we treat all models as though they can return multiple tool or function calls in
-each message
+each message.
 :::

 Tool calling allows a model to respond to a given prompt by generating output that
@@ -831,10 +614,8 @@ from unstructured text, you could give the model an "extraction" tool that takes
 parameters matching the desired schema, then treat the generated output as your final
 result.

-For models that support it, tool calling can be very convenient. It removes the
-guesswork around how best to prompt schemas in favor of a built-in model feature. It can also
-more naturally support agentic flows, since you can just pass multiple tool schemas instead
-of fiddling with enums or unions.
+A tool call includes a name, arguments dict, and an optional identifier. The
+arguments dict is structured `{argument_name: argument_value}`.

 Many LLM providers, including [Anthropic](https://www.anthropic.com/),
 [Cohere](https://cohere.com/), [Google](https://cloud.google.com/vertex-ai),
@@ -851,174 +632,40 @@ LangChain provides a standardized interface for tool calling that is consistent

 The standard interface consists of:

-* `ChatModel.bind_tools()`: a method for specifying which tools are available for a model to call. This method accepts [LangChain tools](/docs/concepts/#tools) here.
+* `ChatModel.bind_tools()`: a method for specifying which tools are available for a model to call.
 * `AIMessage.tool_calls`: an attribute on the `AIMessage` returned from the model for accessing the tool calls requested by the model.

-The following how-to guides are good practical resources for using function/tool calling:
+There are two main use cases for function/tool calling:

 - [How to return structured data from an LLM](/docs/how_to/structured_output/)
 - [How to use a model to call tools](/docs/how_to/tool_calling/)

-For a full list of model providers that support tool calling, [see this table](/docs/integrations/chat/#advanced-features).
-
 ### Retrieval

-LLMs are trained on a large but fixed dataset, limiting their ability to reason over private or recent information. Fine-tuning an LLM with specific facts is one way to mitigate this, but is often [poorly suited for factual recall](https://www.anyscale.com/blog/fine-tuning-is-for-form-not-facts) and [can be costly](https://www.glean.com/blog/how-to-build-an-ai-assistant-for-the-enterprise). 
-Retrieval is the process of providing relevant information to an LLM to improve its response for a given input. Retrieval augmented generation (RAG) is the process of grounding the LLM generation (output) using the retrieved information.
+LangChain provides several advanced retrieval types. A full list is below, along with the following information:

-:::tip
+**Name**: Name of the retrieval algorithm.

-* See our RAG from Scratch [code](https://github.com/langchain-ai/rag-from-scratch) and [video series](https://youtube.com/playlist?list=PLfaIDFEXuae2LXbO1_PKyVJiQ23ZztA0x&feature=shared).
-* For a high-level guide on retrieval, see this [tutorial on RAG](/docs/tutorials/rag/).
+**Index Type**: Which index type (if any) this relies on.

-:::
+**Uses an LLM**: Whether this retrieval method uses an LLM.

-RAG is only as good as the retrieved documents’ relevance and quality. Fortunately, an emerging set of techniques can be employed to design and improve RAG systems. We've focused on taxonomizing and summarizing many of these techniques (see below figure) and will share some high-level strategic guidance in the following sections.
-You can and should experiment with using different pieces together. You might also find [this LangSmith guide](https://docs.smith.langchain.com/how_to_guides/evaluation/evaluate_llm_application) useful for showing how to evaluate different iterations of your app.
+**When to Use**: Our commentary on when you should considering using this retrieval method.

-![](/img/rag_landscape.png)
-
-#### Query Translation
-
-First, consider the user input(s) to your RAG system. Ideally, a RAG system can handle a wide range of inputs, from poorly worded questions to complex multi-part queries.
-**Using an LLM to review and optionally modify the input is the central idea behind query translation.** This serves as a general buffer, optimizing raw user inputs for your retrieval system. 
-For example, this can be as simple as extracting keywords or as complex as generating multiple sub-questions for a complex query.
-
-| Name          | When to use | Description |
-|---------------|-------------|-------------|
-| [Multi-query](/docs/how_to/MultiQueryRetriever/)   | When you need to cover multiple perspectives of a question. | Rewrite the user question from multiple perspectives, retrieve documents for each rewritten question, return the unique documents for all queries. |
-| [Decomposition](https://github.com/langchain-ai/rag-from-scratch/blob/main/rag_from_scratch_5_to_9.ipynb) | When a question can be broken down into smaller subproblems. | Decompose a question into a set of subproblems / questions, which can either be solved sequentially (use the answer from first + retrieval to answer the second) or in parallel (consolidate each answer into final answer). |
-| [Step-back](https://github.com/langchain-ai/rag-from-scratch/blob/main/rag_from_scratch_5_to_9.ipynb)     | When a higher-level conceptual understanding is required. | First prompt the LLM to ask a generic step-back question about higher-level concepts or principles, and retrieve relevant facts about them. Use this grounding to help answer the user question. |
-| [HyDE](https://github.com/langchain-ai/rag-from-scratch/blob/main/rag_from_scratch_5_to_9.ipynb)          | If you have challenges retrieving relevant documents using the raw user inputs. | Use an LLM to convert questions into hypothetical documents that answer the question. Use the embedded hypothetical documents to retrieve real documents with the premise that doc-doc similarity search can produce more relevant matches. |
-
-:::tip
-
-See our RAG from Scratch videos for a few different specific approaches:
- [Multi-query](https://youtu.be/JChPi0CRnDY?feature=shared)
- [Decomposition](https://youtu.be/h0OPWlEOank?feature=shared)
- [Step-back](https://youtu.be/xn1jEjRyJ2U?feature=shared)
- [HyDE](https://youtu.be/SaDzIVkYqyY?feature=shared)
-
-:::
-
-#### Routing
-
-Second, consider the data sources available to your RAG system. You want to query across more than one database or across structured and unstructured data sources. **Using an LLM to review the input and route it to the appropriate data source is a simple and effective approach for querying across sources.**
-
-| Name             | When to use                                | Description |
-|------------------|--------------------------------------------|-------------|
-| [Logical routing](/docs/how_to/routing/)  | When you can prompt an LLM with rules to decide where to route the input. | Logical routing can use an LLM to reason about the query and choose which datastore is most appropriate. |
-| [Semantic routing](/docs/how_to/routing/#routing-by-semantic-similarity) | When semantic similarity is an effective way to determine where to route the input. | Semantic routing embeds both query and, typically a set of prompts. It then chooses the appropriate prompt based upon similarity. |
-
-:::tip
-
-See our RAG from Scratch video on [routing](https://youtu.be/pfpIndq7Fi8?feature=shared).  
-
-:::
-
-#### Query Construction
-
-Third, consider whether any of your data sources require specific query formats. Many structured databases use SQL. Vector stores often have specific syntax for applying keyword filters to document metadata. **Using an LLM to convert a natural language query into a query syntax is a popular and powerful approach.**
-In particular, [text-to-SQL](/docs/tutorials/sql_qa/), [text-to-Cypher](/docs/tutorials/graph/), and [query analysis for metadata filters](/docs/tutorials/query_analysis/#query-analysis) are useful ways to interact with structured, graph, and vector databases respectively. 
-
-| Name                                        | When to Use                                                                                                                                   | Description                                                                                                                                                                                                                                                                                      |
-|---------------------------------------------|-----------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
-| [Text to SQL](/docs/tutorials/sql_qa/)      | If users are asking questions that require information housed in a relational database, accessible via SQL.                                   | This uses an LLM to transform user input into a SQL query.                                             |
-| [Text-to-Cypher](/docs/tutorials/graph/)    | If users are asking questions that require information housed in a graph database, accessible via Cypher.                                     | This uses an LLM to transform user input into a Cypher query.                                              |
-| [Self Query](/docs/how_to/self_query/)      | If users are asking questions that are better answered by fetching documents based on metadata rather than similarity with the text.          | This uses an LLM to transform user input into two things: (1) a string to look up semantically, (2) a metadata filter to go along with it. This is useful because oftentimes questions are about the METADATA of documents (not the content itself).                                              |
-
-:::tip
-
-See our [blog post overview](https://blog.langchain.dev/query-construction/) and RAG from Scratch video on [query construction](https://youtu.be/kl6NwWYxvbM?feature=shared), the process of text-to-DSL where DSL is a domain specific language required to interact with a given database. This converts user questions into structured queries. 
-
-:::
-
-#### Indexing
-
-Fouth, consider the design of your document index. A simple and powerful idea is to **decouple the documents that you index for retrieval from the documents that you pass to the LLM for generation.** Indexing frequently uses embedding models with vector stores, which [compress the semantic information in documents to fixed-size vectors](/docs/concepts/#embedding-models).
-
-Many RAG approaches focus on splitting documents into chunks and retrieving some number based on similarity to an input question for the LLM. But chunk size and chunk number can be difficult to set and affect results if they do not provide full context for the LLM to answer a question. Furthermore, LLMs are increasingly capable of processing millions of tokens. 
-
-Two approaches can address this tension: (1) [Multi Vector](/docs/how_to/multi_vector/) retriever using an LLM to translate documents into any form (e.g., often into a summary) that is well-suited for indexing, but returns full documents to the LLM for generation. (2) [ParentDocument](/docs/how_to/parent_document_retriever/) retriever embeds document chunks, but also returns full documents. The idea is to get the best of both worlds: use concise representations (summaries or chunks) for retrieval, but use the full documents for answer generation.
-
-| Name                      | Index Type                   | Uses an LLM               | When to Use                                                                                                                                   | Description                                                                                                                                                                                                                                                                                      |
-|---------------------------|------------------------------|---------------------------|-----------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
-| [Vector store](/docs/how_to/vectorstore_retriever/)               | Vector store                  | No                        | If you are just getting started and looking for something quick and easy.                                                                     | This is the simplest method and the one that is easiest to get started with. It involves creating embeddings for each piece of text.                                                                                                                                                             |
-| [ParentDocument](/docs/how_to/parent_document_retriever/)            | Vector store + Document Store | No                        | If your pages have lots of smaller pieces of distinct information that are best indexed by themselves, but best retrieved all together.       | This involves indexing multiple chunks for each document. Then you find the chunks that are most similar in embedding space, but you retrieve the whole parent document and return that (rather than individual chunks).                                                                         |
-| [Multi Vector](/docs/how_to/multi_vector/)              | Vector store + Document Store | Sometimes during indexing | If you are able to extract information from documents that you think is more relevant to index than the text itself.                          | This involves creating multiple vectors for each document. Each vector could be created in a myriad of ways - examples include summaries of the text and hypothetical questions.                                                                                                                 |
-| [Time-Weighted Vector store](/docs/how_to/time_weighted_vectorstore/) | Vector store                  | No                        | If you have timestamps associated with your documents, and you want to retrieve the most recent ones                                          | This fetches documents based on a combination of semantic similarity (as in normal vector retrieval) and recency (looking at timestamps of indexed documents)                                                                                                                                    |
-
-:::tip
-
- See our RAG from Scratch video on [indexing fundamentals](https://youtu.be/bjb_EMsTDKI?feature=shared)
- See our RAG from Scratch video on [multi vector retriever](https://youtu.be/gTCU9I6QqCE?feature=shared)
-
-:::
-
-Fifth, consider ways to improve the quality of your similarity search itself. Embedding models compress text into fixed-length (vector) representations that capture the semantic content of the document. This compression is useful for search / retrieval, but puts a heavy burden on that single vector representation to capture the semantic nuance / detail of the document. In some cases, irrelevant or redundant content can dilute the semantic usefulness of the embedding.
-
-[ColBERT](https://docs.google.com/presentation/d/1IRhAdGjIevrrotdplHNcc4aXgIYyKamUKTWtB3m3aMU/edit?usp=sharing) is an interesting approach to address this with a higher granularity embeddings: (1) produce a contextually influenced embedding for each token in the document and query, (2) score similarity between each query token and all document tokens, (3) take the max, (4) do this for all query tokens, and (5) take the sum of the max scores (in step 3) for all query tokens to get a query-document similarity score; this token-wise scoring can yield strong results. 
-
-![](/img/colbert.png)
-
-There are some additional tricks to improve the quality of your retrieval. Embeddings excel at capturing semantic information, but may struggle with keyword-based queries. Many [vector stores](/docs/integrations/retrievers/pinecone_hybrid_search/) offer built-in [hybrid-search](https://docs.pinecone.io/guides/data/understanding-hybrid-search) to combine keyword and semantic similarity, which marries the benefits of both approaches. Furthermore, many vector stores have [maximal marginal relevance](https://python.langchain.com/v0.1/docs/modules/model_io/prompts/example_selectors/mmr/), which attempts to diversify the results of a search to avoid returning similar and redundant documents. 
-
-| Name              | When to use                                              | Description |
-|-------------------|----------------------------------------------------------|-------------|
-| [ColBERT](/docs/integrations/providers/ragatouille/#using-colbert-as-a-reranker)           | When higher granularity embeddings are needed.           | ColBERT uses contextually influenced embeddings for each token in the document and query to get a granular query-document similarity score. |
-| [Hybrid search](/docs/integrations/retrievers/pinecone_hybrid_search/)     | When combining keyword-based and semantic similarity.    | Hybrid search combines keyword and semantic similarity, marrying the benefits of both approaches. |
-| [Maximal Marginal Relevance (MMR)](/docs/integrations/vectorstores/pinecone/#maximal-marginal-relevance-searches) | When needing to diversify search results. | MMR attempts to diversify the results of a search to avoid returning similar and redundant documents. |
-
-:::tip
-
-See our RAG from Scratch video on [ColBERT](https://youtu.be/cN6S0Ehm7_8?feature=shared>).
-
-:::
-
-#### Post-processing
-
-Sixth, consider ways to filter or rank retrieved documents. This is very useful if you are [combining documents returned from multiple sources](/docs/integrations/retrievers/cohere-reranker/#doing-reranking-with-coherererank), since it can can down-rank less relevant documents and / or [compress similar documents](/docs/how_to/contextual_compression/#more-built-in-compressors-filters). 
+**Description**: Description of what this retrieval algorithm is doing.

 | Name                      | Index Type                   | Uses an LLM               | When to Use                                                                                                                                   | Description                                                                                                                                                                                                                                                                                      |
 |---------------------------|------------------------------|---------------------------|-----------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
+| [Vectorstore](/docs/how_to/vectorstore_retriever/)               | Vectorstore                  | No                        | If you are just getting started and looking for something quick and easy.                                                                     | This is the simplest method and the one that is easiest to get started with. It involves creating embeddings for each piece of text.                                                                                                                                                             |
+| [ParentDocument](/docs/how_to/parent_document_retriever/)            | Vectorstore + Document Store | No                        | If your pages have lots of smaller pieces of distinct information that are best indexed by themselves, but best retrieved all together.       | This involves indexing multiple chunks for each document. Then you find the chunks that are most similar in embedding space, but you retrieve the whole parent document and return that (rather than individual chunks).                                                                         |
+| [Multi Vector](/docs/how_to/multi_vector/)              | Vectorstore + Document Store | Sometimes during indexing | If you are able to extract information from documents that you think is more relevant to index than the text itself.                          | This involves creating multiple vectors for each document. Each vector could be created in a myriad of ways - examples include summaries of the text and hypothetical questions.                                                                                                                 |
+| [Self Query](/docs/how_to/self_query/)               | Vectorstore                  | Yes                       | If users are asking questions that are better answered by fetching documents based on metadata rather than similarity with the text.          | This uses an LLM to transform user input into two things: (1) a string to look up semantically, (2) a metadata filer to go along with it. This is useful because oftentimes questions are about the METADATA of documents (not the content itself).                                              |
 | [Contextual Compression](/docs/how_to/contextual_compression/)    | Any                          | Sometimes                 | If you are finding that your retrieved documents contain too much irrelevant information and are distracting the LLM.                         | This puts a post-processing step on top of another retriever and extracts only the most relevant information from retrieved documents. This can be done with embeddings or an LLM.                                                                                                               |
+| [Time-Weighted Vectorstore](/docs/how_to/time_weighted_vectorstore/) | Vectorstore                  | No                        | If you have timestamps associated with your documents, and you want to retrieve the most recent ones                                          | This fetches documents based on a combination of semantic similarity (as in normal vector retrieval) and recency (looking at timestamps of indexed documents)                                                                                                                                    |
+| [Multi-Query Retriever](/docs/how_to/MultiQueryRetriever/)     | Any                          | Yes                       | If users are asking questions that are complex and require multiple pieces of distinct information to respond                                 | This uses an LLM to generate multiple queries from the original one. This is useful when the original query needs pieces of information about multiple topics to be properly answered. By generating multiple queries, we can then fetch documents for each of them.                             |
 | [Ensemble](/docs/how_to/ensemble_retriever/)                  | Any                          | No                        | If you have multiple retrieval methods and want to try combining them.                                                                        | This fetches documents from multiple retrievers and then combines them.                                                                                                                                                                                                                          |
-| [Re-ranking](/docs/integrations/retrievers/cohere-reranker/)                  | Any                          | Yes                        |  If you want to rank retrieved documents based upon relevance, especially if you want to combine results from multiple retrieval methods .                                                                         | Given a query and a list of documents, Rerank indexes the documents from most to least semantically relevant to the query.                                                                                                                                                                                                                         |

-:::tip
-
-See our RAG from Scratch video on [RAG-Fusion](https://youtu.be/77qELPbNgxA?feature=shared), on approach for post-processing across multiple queries:  Rewrite the user question from multiple perspectives, retrieve documents for each rewritten question, and combine the ranks of multiple search result lists to produce a single, unified ranking with [Reciprocal Rank Fusion (RRF)](https://towardsdatascience.com/forget-rag-the-future-is-rag-fusion-1147298d8ad1).
-
-:::
-
-#### Generation
-
-**Finally, consider ways to build self-correction into your RAG system.** RAG systems can suffer from low quality retrieval (e.g., if a user question is out of the domain for the index) and / or hallucinations in generation. A naive retrieve-generate pipeline has no ability to detect or self-correct from these kinds of errors. The concept of ["flow engineering"](https://x.com/karpathy/status/1748043513156272416) has been introduced [in the context of code generation](https://arxiv.org/abs/2401.08500): iteratively build an answer to a code question with unit tests to check and self-correct errors. Several works have applied this RAG, such as Self-RAG and Corrective-RAG. In both cases, checks for document relevance, hallucinations, and / or answer quality are performed in the RAG answer generation flow.
-
-We've found that graphs are a great way to reliably express logical flows and have implemented ideas from several of these papers [using LangGraph](https://github.com/langchain-ai/langgraph/tree/main/examples/rag), as shown in the figure below (red - routing, blue - fallback, green - self-correction):
- **Routing:** Adaptive RAG ([paper](https://arxiv.org/abs/2403.14403)). Route questions to different retrieval approaches, as discussed above 
- **Fallback:** Corrective RAG ([paper](https://arxiv.org/pdf/2401.15884.pdf)). Fallback to web search if docs are not relevant to query
- **Self-correction:** Self-RAG ([paper](https://arxiv.org/abs/2310.11511)). Fix answers w/ hallucinations or don’t address question
-
-![](/img/langgraph_rag.png)
-
-| Name              | When to use                                               | Description |
-|-------------------|-----------------------------------------------------------|-------------|
-| Self-RAG          | When needing to fix answers with hallucinations or irrelevant content. | Self-RAG performs checks for document relevance, hallucinations, and answer quality during the RAG answer generation flow, iteratively building an answer and self-correcting errors. |
-| Corrective-RAG    | When needing a fallback mechanism for low relevance docs. | Corrective-RAG includes a fallback (e.g., to web search) if the retrieved documents are not relevant to the query, ensuring higher quality and more relevant retrieval. |
-
-:::tip
-
-See several videos and cookbooks showcasing RAG with LangGraph: 
- [LangGraph Corrective RAG](https://www.youtube.com/watch?v=E2shqsYwxck)
- [LangGraph combining Adaptive, Self-RAG, and Corrective RAG](https://www.youtube.com/watch?v=-ROS6gfYIts) 
- [Cookbooks for RAG using LangGraph](https://github.com/langchain-ai/langgraph/tree/main/examples/rag)
-
-See our LangGraph RAG recipes with partners:
- [Meta](https://github.com/meta-llama/llama-recipes/tree/main/recipes/use_cases/agents/langchain) 
- [Mistral](https://github.com/mistralai/cookbook/tree/main/third_party/langchain)
-
-:::
+For a high-level guide on retrieval, see this [tutorial on RAG](/docs/tutorials/rag/).

 ### Text splitting

@@ -1045,19 +692,6 @@ Table columns:
 | Semantic Chunker (Experimental) | [SemanticChunker](/docs/how_to/semantic-chunker/)                                                                                                                             | Sentences                                                                                                       |               | First splits on sentences. Then combines ones next to each other if they are semantically similar enough. Taken from [Greg Kamradt](https://github.com/FullStackRetrieval-com/RetrievalTutorials/blob/main/tutorials/LevelsOfTextSplitting/5_Levels_Of_Text_Splitting.ipynb) |
 | Integration: AI21 Semantic | [AI21SemanticTextSplitter](/docs/integrations/document_transformers/ai21_semantic_text_splitter/)                                                                                                                    |    ✅           | Identifies distinct topics that form coherent pieces of text and splits along those.                                                                                                                                                                                         |

-### Evaluation
-<span data-heading-keywords="evaluation,evaluate"></span>

-Evaluation is the process of assessing the performance and effectiveness of your LLM-powered applications.
-It involves testing the model's responses against a set of predefined criteria or benchmarks to ensure it meets the desired quality standards and fulfills the intended purpose.
-This process is vital for building reliable applications.

-![](/img/langsmith_evaluate.png)

-[LangSmith](https://docs.smith.langchain.com/) helps with this process in a few ways:
-
- It makes it easier to create and curate datasets via its tracing and annotation features
- It provides an evaluation framework that helps you define metrics and run your app against your dataset
- It allows you to track results over time and automatically run your evaluators on a schedule or as part of CI/Code
-
-To learn more, check out [this LangSmith guide](https://docs.smith.langchain.com/concepts/evaluation).
--- a/docs/docs/contributing/documentation/style_guide.mdx
+++ b/docs/docs/contributing/documentation/style_guide.mdx
@@ -55,7 +55,7 @@ The below sections are listed roughly in order of increasing level of abstractio

 ### Expression Language

-[LangChain Expression Language (LCEL)](/docs/concepts#langchain-expression-language-lcel) is the fundamental way that most LangChain components fit together, and this section is designed to teach
+[LangChain Expression Language (LCEL)](/docs/concepts#langchain-expression-language) is the fundamental way that most LangChain components fit together, and this section is designed to teach
 developers how to use it to build with LangChain's primitives effectively.

 This section should contains **Tutorials** that teach how to stream and use LCEL primitives for more abstract tasks, **Explanations** of specific behaviors,
--- a/docs/docs/contributing/index.mdx
+++ b/docs/docs/contributing/index.mdx
@@ -48,7 +48,7 @@ In a similar vein, we do enforce certain linting, formatting, and documentation
 If you are finding these difficult (or even just annoying) to work with, feel free to contact a maintainer for help -
 we do not want these to get in the way of getting good code into the codebase.

-### 🌟 Recognition
+# 🌟 Recognition

 If your contribution has made its way into a release, we will want to give you credit on Twitter (only if you want though)!
-If you have a Twitter account you would like us to mention, please let us know in the PR or through another means.
+If you have a Twitter account you would like us to mention, please let us know in the PR or through another means.
--- a/docs/docs/contributing/repo_structure.mdx
+++ b/docs/docs/contributing/repo_structure.mdx
@@ -15,22 +15,12 @@ Here's the structure visualized as a tree:
 ├── cookbook # Tutorials and examples
 ├── docs # Contains content for the documentation here: https://python.langchain.com/
 ├── libs
-│   ├── langchain
-│   │   ├── langchain
+│   ├── langchain # Main package
 │   │   ├── tests/unit_tests # Unit tests (present in each package not shown for brevity)
 │   │   ├── tests/integration_tests # Integration tests (present in each package not shown for brevity)
-│   ├── community # Third-party integrations
-│   │   ├── langchain-community
-│   ├── core # Base interfaces for key abstractions
-│   │   ├── langchain-core
-│   ├── experimental # Experimental components and chains
-│   │   ├── langchain-experimental
-|   ├── cli # Command line interface
-│   │   ├── langchain-cli
-│   ├── text-splitters
-│   │   ├── langchain-text-splitters
-│   ├── standard-tests
-│   │   ├── langchain-standard-tests
+│   ├── langchain-community # Third-party integrations
+│   ├── langchain-core # Base interfaces for key abstractions
+│   ├── langchain-experimental # Experimental components and chains
 │   ├── partners
 │       ├── langchain-partner-1
 │       ├── langchain-partner-2
--- a/docs/docs/how_to/chat_models_universal_init.ipynb
+++ b/docs/docs/how_to/chat_models_universal_init.ipynb
@@ -5,7 +5,7 @@
   "id": "cfdf4f09-8125-4ed1-8063-6feed57da8a3",
   "metadata": {},
   "source": [
-    "# How to init any model in one line\n",
+    "# How to let your end users choose their model\n",
    "\n",
    "Many LLM applications let end users specify what model provider and model they want the application to be powered by. This requires writing some logic to initialize different ChatModels based on some user configuration. The `init_chat_model()` helper method makes it easy to initialize a number of different model integrations without having to worry about import paths and class names.\n",
    "\n",
--- a/docs/docs/how_to/filter_messages.ipynb
+++ b/docs/docs/how_to/filter_messages.ipynb
@@ -1,203 +0,0 @@
-{
- "cells": [
-  {
-   "cell_type": "markdown",
-   "id": "e389175d-8a65-4f0d-891c-dbdfabb3c3ef",
-   "metadata": {},
-   "source": [
-    "# How to filter messages\n",
-    "\n",
-    "In more complex chains and agents we might track state with a list of messages. This list can start to accumulate messages from multiple different models, speakers, sub-chains, etc., and we may only want to pass subsets of this full list of messages to each model call in the chain/agent.\n",
-    "\n",
-    "The `filter_messages` utility makes it easy to filter messages by type, id, or name.\n",
-    "\n",
-    "## Basic usage"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 1,
-   "id": "f4ad2fd3-3cab-40d4-a989-972115865b8b",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[HumanMessage(content='example input', name='example_user', id='2'),\n",
-       " HumanMessage(content='real input', name='bob', id='4')]"
-      ]
-     },
-     "execution_count": 1,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "from langchain_core.messages import (\n",
-    "    AIMessage,\n",
-    "    HumanMessage,\n",
-    "    SystemMessage,\n",
-    "    filter_messages,\n",
-    ")\n",
-    "\n",
-    "messages = [\n",
-    "    SystemMessage(\"you are a good assistant\", id=\"1\"),\n",
-    "    HumanMessage(\"example input\", id=\"2\", name=\"example_user\"),\n",
-    "    AIMessage(\"example output\", id=\"3\", name=\"example_assistant\"),\n",
-    "    HumanMessage(\"real input\", id=\"4\", name=\"bob\"),\n",
-    "    AIMessage(\"real output\", id=\"5\", name=\"alice\"),\n",
-    "]\n",
-    "\n",
-    "filter_messages(messages, include_types=\"human\")"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 2,
-   "id": "7b663a1e-a8ae-453e-a072-8dd75dfab460",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[SystemMessage(content='you are a good assistant', id='1'),\n",
-       " HumanMessage(content='real input', name='bob', id='4'),\n",
-       " AIMessage(content='real output', name='alice', id='5')]"
-      ]
-     },
-     "execution_count": 2,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "filter_messages(messages, exclude_names=[\"example_user\", \"example_assistant\"])"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 3,
-   "id": "db170e46-03f8-4710-b967-23c70c3ac054",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[HumanMessage(content='example input', name='example_user', id='2'),\n",
-       " HumanMessage(content='real input', name='bob', id='4'),\n",
-       " AIMessage(content='real output', name='alice', id='5')]"
-      ]
-     },
-     "execution_count": 3,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "filter_messages(messages, include_types=[HumanMessage, AIMessage], exclude_ids=[\"3\"])"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "b7c4e5ad-d1b4-4c18-b250-864adde8f0dd",
-   "metadata": {},
-   "source": [
-    "## Chaining\n",
-    "\n",
-    "`filter_messages` can be used in an imperatively (like above) or declaratively, making it easy to compose with other components in a chain:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 4,
-   "id": "675f8f79-db39-401c-a582-1df2478cba30",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "AIMessage(content=[], response_metadata={'id': 'msg_01Wz7gBHahAwkZ1KCBNtXmwA', 'model': 'claude-3-sonnet-20240229', 'stop_reason': 'end_turn', 'stop_sequence': None, 'usage': {'input_tokens': 16, 'output_tokens': 3}}, id='run-b5d8a3fe-004f-4502-a071-a6c025031827-0', usage_metadata={'input_tokens': 16, 'output_tokens': 3, 'total_tokens': 19})"
-      ]
-     },
-     "execution_count": 4,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "# pip install -U langchain-anthropic\n",
-    "from langchain_anthropic import ChatAnthropic\n",
-    "\n",
-    "llm = ChatAnthropic(model=\"claude-3-sonnet-20240229\", temperature=0)\n",
-    "# Notice we don't pass in messages. This creates\n",
-    "# a RunnableLambda that takes messages as input\n",
-    "filter_ = filter_messages(exclude_names=[\"example_user\", \"example_assistant\"])\n",
-    "chain = filter_ | llm\n",
-    "chain.invoke(messages)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "4133ab28-f49c-480f-be92-b51eb6559153",
-   "metadata": {},
-   "source": [
-    "Looking at the LangSmith trace we can see that before the messages are passed to the model they are filtered: https://smith.langchain.com/public/f808a724-e072-438e-9991-657cc9e7e253/r\n",
-    "\n",
-    "Looking at just the filter_, we can see that it's a Runnable object that can be invoked like all Runnables:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 6,
-   "id": "c090116a-1fef-43f6-a178-7265dff9db00",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[HumanMessage(content='real input', name='bob', id='4'),\n",
-       " AIMessage(content='real output', name='alice', id='5')]"
-      ]
-     },
-     "execution_count": 6,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "filter_.invoke(messages)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "ff339066-d424-4042-8cca-cd4b007c1a8e",
-   "metadata": {},
-   "source": [
-    "## API reference\n",
-    "\n",
-    "For a complete description of all arguments head to the API reference: https://api.python.langchain.com/en/latest/messages/langchain_core.messages.utils.filter_messages.html"
-   ]
-  }
- ],
- "metadata": {
-  "kernelspec": {
-   "display_name": "poetry-venv-2",
-   "language": "python",
-   "name": "poetry-venv-2"
-  },
-  "language_info": {
-   "codemirror_mode": {
-    "name": "ipython",
-    "version": 3
-   },
-   "file_extension": ".py",
-   "mimetype": "text/x-python",
-   "name": "python",
-   "nbconvert_exporter": "python",
-   "pygments_lexer": "ipython3",
-   "version": "3.9.1"
-  }
- },
- "nbformat": 4,
- "nbformat_minor": 5
-}
--- a/docs/docs/how_to/index.mdx
+++ b/docs/docs/how_to/index.mdx
@@ -14,7 +14,6 @@ For comprehensive descriptions of every class and function see the [API Referenc
 ## Installation

 - [How to: install LangChain packages](/docs/how_to/installation/)
- [How to: use LangChain with different Pydantic versions](/docs/how_to/pydantic_compatibility)

 ## Key features

@@ -79,15 +78,7 @@ These are the core building blocks you can use when building applications.
 - [How to: stream a response back](/docs/how_to/chat_streaming)
 - [How to: track token usage](/docs/how_to/chat_token_usage_tracking)
 - [How to: track response metadata across providers](/docs/how_to/response_metadata)
- [How to: init any model in one line](/docs/how_to/chat_models_universal_init/)
-
-### Messages
-
-[Messages](/docs/concepts/#messages) are the input and output of chat models. They have some `content` and a `role`, which describes the source of the message.
-
- [How to: trim messages](/docs/how_to/trim_messages/)
- [How to: filter messages](/docs/how_to/filter_messages/)
- [How to: merge consecutive messages of the same type](/docs/how_to/merge_message_runs/)
+- [How to: let your end users choose their model](/docs/how_to/chat_models_universal_init/)

 ### LLMs

@@ -259,7 +250,6 @@ For a high-level tutorial on building chatbots, check out [this guide](/docs/tut
 - [How to: manage memory](/docs/how_to/chatbots_memory)
 - [How to: do retrieval](/docs/how_to/chatbots_retrieval)
 - [How to: use tools](/docs/how_to/chatbots_tools)
- [How to: manage large chat history](/docs/how_to/trim_messages/)

 ### Query analysis

@@ -304,15 +294,7 @@ You can peruse [LangGraph how-to guides here](https://langchain-ai.github.io/lan
 ## [LangSmith](https://docs.smith.langchain.com/)

 LangSmith allows you to closely trace, monitor and evaluate your LLM application.
-It seamlessly integrates with LangChain and LangGraph, and you can use it to inspect and debug individual steps of your chains and agents as you build.
+It seamlessly integrates with LangChain, and you can use it to inspect and debug individual steps of your chains as you build.

 LangSmith documentation is hosted on a separate site.
 You can peruse [LangSmith how-to guides here](https://docs.smith.langchain.com/how_to_guides/).
-
-### Evaluation
-<span data-heading-keywords="evaluation,evaluate"></span>
-
-Evaluating performance is a vital part of building LLM-powered applications.
-LangSmith helps with every step of the process from creating a dataset to defining metrics to running evaluators.
-
-To learn more, check out the [LangSmith evaluation how-to guides](https://docs.smith.langchain.com/how_to_guides/evaluation).
--- a/docs/docs/how_to/indexing.ipynb
+++ b/docs/docs/how_to/indexing.ipynb
@@ -60,7 +60,7 @@
    "   * document addition by id (`add_documents` method with `ids` argument)\n",
    "   * delete by id (`delete` method with `ids` argument)\n",
    "\n",
-    "Compatible Vectorstores: `Aerospike`, `AnalyticDB`, `AstraDB`, `AwaDB`, `AzureCosmosDBNoSqlVectorSearch`, `AzureCosmosDBVectorSearch`, `Bagel`, `Cassandra`, `Chroma`, `CouchbaseVectorStore`, `DashVector`, `DatabricksVectorSearch`, `DeepLake`, `Dingo`, `ElasticVectorSearch`, `ElasticsearchStore`, `FAISS`, `HanaDB`, `Milvus`, `MyScale`, `OpenSearchVectorSearch`, `PGVector`, `Pinecone`, `Qdrant`, `Redis`, `Rockset`, `ScaNN`, `SupabaseVectorStore`, `SurrealDBStore`, `TimescaleVector`, `Vald`, `VDMS`, `Vearch`, `VespaStore`, `Weaviate`, `Yellowbrick`, `ZepVectorStore`, `TencentVectorDB`, `OpenSearchVectorSearch`.\n",
+    "Compatible Vectorstores: `Aerospike`, `AnalyticDB`, `AstraDB`, `AwaDB`, `Bagel`, `Cassandra`, `Chroma`, `CouchbaseVectorStore`, `DashVector`, `DatabricksVectorSearch`, `DeepLake`, `Dingo`, `ElasticVectorSearch`, `ElasticsearchStore`, `FAISS`, `HanaDB`, `Milvus`, `MyScale`, `OpenSearchVectorSearch`, `PGVector`, `Pinecone`, `Qdrant`, `Redis`, `Rockset`, `ScaNN`, `SupabaseVectorStore`, `SurrealDBStore`, `TimescaleVector`, `Vald`, `VDMS`, `Vearch`, `VespaStore`, `Weaviate`, `Yellowbrick`, `ZepVectorStore`, `TencentVectorDB`, `OpenSearchVectorSearch`.\n",
    "  \n",
    "## Caution\n",
    "\n",
--- a/docs/docs/how_to/merge_message_runs.ipynb
+++ b/docs/docs/how_to/merge_message_runs.ipynb
@@ -1,170 +0,0 @@
-{
- "cells": [
-  {
-   "cell_type": "markdown",
-   "id": "ac47bfab-0f4f-42ce-8bb6-898ef22a0338",
-   "metadata": {},
-   "source": [
-    "# How to merge consecutive messages of the same type\n",
-    "\n",
-    "Certain models do not support passing in consecutive messages of the same type (a.k.a. \"runs\" of the same message type).\n",
-    "\n",
-    "The `merge_message_runs` utility makes it easy to merge consecutive messages of the same type.\n",
-    "\n",
-    "## Basic usage"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 1,
-   "id": "1a215bbb-c05c-40b0-a6fd-d94884d517df",
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "SystemMessage(content=\"you're a good assistant.\\nyou always respond with a joke.\")\n",
-      "\n",
-      "HumanMessage(content=[{'type': 'text', 'text': \"i wonder why it's called langchain\"}, 'and who is harrison chasing anyways'])\n",
-      "\n",
-      "AIMessage(content='Well, I guess they thought \"WordRope\" and \"SentenceString\" just didn\\'t have the same ring to it!\\nWhy, he\\'s probably chasing after the last cup of coffee in the office!')\n"
-     ]
-    }
-   ],
-   "source": [
-    "from langchain_core.messages import (\n",
-    "    AIMessage,\n",
-    "    HumanMessage,\n",
-    "    SystemMessage,\n",
-    "    merge_message_runs,\n",
-    ")\n",
-    "\n",
-    "messages = [\n",
-    "    SystemMessage(\"you're a good assistant.\"),\n",
-    "    SystemMessage(\"you always respond with a joke.\"),\n",
-    "    HumanMessage([{\"type\": \"text\", \"text\": \"i wonder why it's called langchain\"}]),\n",
-    "    HumanMessage(\"and who is harrison chasing anyways\"),\n",
-    "    AIMessage(\n",
-    "        'Well, I guess they thought \"WordRope\" and \"SentenceString\" just didn\\'t have the same ring to it!'\n",
-    "    ),\n",
-    "    AIMessage(\"Why, he's probably chasing after the last cup of coffee in the office!\"),\n",
-    "]\n",
-    "\n",
-    "merged = merge_message_runs(messages)\n",
-    "print(\"\\n\\n\".join([repr(x) for x in merged]))"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "0544c811-7112-4b76-8877-cc897407c738",
-   "metadata": {},
-   "source": [
-    "Notice that if the contents of one of the messages to merge is a list of content blocks then the merged message will have a list of content blocks. And if both messages to merge have string contents then those are concatenated with a newline character."
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "1b2eee74-71c8-4168-b968-bca580c25d18",
-   "metadata": {},
-   "source": [
-    "## Chaining\n",
-    "\n",
-    "`merge_message_runs` can be used in an imperatively (like above) or declaratively, making it easy to compose with other components in a chain:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 3,
-   "id": "6d5a0283-11f8-435b-b27b-7b18f7693592",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "AIMessage(content=[], response_metadata={'id': 'msg_01D6R8Naum57q8qBau9vLBUX', 'model': 'claude-3-sonnet-20240229', 'stop_reason': 'end_turn', 'stop_sequence': None, 'usage': {'input_tokens': 84, 'output_tokens': 3}}, id='run-ac0c465b-b54f-4b8b-9295-e5951250d653-0', usage_metadata={'input_tokens': 84, 'output_tokens': 3, 'total_tokens': 87})"
-      ]
-     },
-     "execution_count": 3,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "# pip install -U langchain-anthropic\n",
-    "from langchain_anthropic import ChatAnthropic\n",
-    "\n",
-    "llm = ChatAnthropic(model=\"claude-3-sonnet-20240229\", temperature=0)\n",
-    "# Notice we don't pass in messages. This creates\n",
-    "# a RunnableLambda that takes messages as input\n",
-    "merger = merge_message_runs()\n",
-    "chain = merger | llm\n",
-    "chain.invoke(messages)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "72e90dce-693c-4842-9526-ce6460fe956b",
-   "metadata": {},
-   "source": [
-    "Looking at the LangSmith trace we can see that before the messages are passed to the model they are merged: https://smith.langchain.com/public/ab558677-cac9-4c59-9066-1ecce5bcd87c/r\n",
-    "\n",
-    "Looking at just the merger, we can see that it's a Runnable object that can be invoked like all Runnables:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 4,
-   "id": "460817a6-c327-429d-958e-181a8c46059c",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[SystemMessage(content=\"you're a good assistant.\\nyou always respond with a joke.\"),\n",
-       " HumanMessage(content=[{'type': 'text', 'text': \"i wonder why it's called langchain\"}, 'and who is harrison chasing anyways']),\n",
-       " AIMessage(content='Well, I guess they thought \"WordRope\" and \"SentenceString\" just didn\\'t have the same ring to it!\\nWhy, he\\'s probably chasing after the last cup of coffee in the office!')]"
-      ]
-     },
-     "execution_count": 4,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "merger.invoke(messages)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "4548d916-ce21-4dc6-8f19-eedb8003ace6",
-   "metadata": {},
-   "source": [
-    "## API reference\n",
-    "\n",
-    "For a complete description of all arguments head to the API reference: https://api.python.langchain.com/en/latest/messages/langchain_core.messages.utils.merge_message_runs.html"
-   ]
-  }
- ],
- "metadata": {
-  "kernelspec": {
-   "display_name": "poetry-venv-2",
-   "language": "python",
-   "name": "poetry-venv-2"
-  },
-  "language_info": {
-   "codemirror_mode": {
-    "name": "ipython",
-    "version": 3
-   },
-   "file_extension": ".py",
-   "mimetype": "text/x-python",
-   "name": "python",
-   "nbconvert_exporter": "python",
-   "pygments_lexer": "ipython3",
-   "version": "3.9.1"
-  }
- },
- "nbformat": 4,
- "nbformat_minor": 5
-}
--- a/docs/docs/how_to/message_history.ipynb
+++ b/docs/docs/how_to/message_history.ipynb
@@ -898,7 +898,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
-   "version": "3.9.1"
+   "version": "3.10.1"
  }
 },
 "nbformat": 4,
--- a/docs/docs/how_to/output_parser_structured.ipynb
+++ b/docs/docs/how_to/output_parser_structured.ipynb
@@ -94,7 +94,7 @@
   "source": [
    "## LCEL\n",
    "\n",
-    "Output parsers implement the [Runnable interface](/docs/concepts#interface), the basic building block of the [LangChain Expression Language (LCEL)](/docs/concepts#langchain-expression-language-lcel). This means they support `invoke`, `ainvoke`, `stream`, `astream`, `batch`, `abatch`, `astream_log` calls.\n",
+    "Output parsers implement the [Runnable interface](/docs/concepts#interface), the basic building block of the [LangChain Expression Language (LCEL)](/docs/concepts#langchain-expression-language). This means they support `invoke`, `ainvoke`, `stream`, `astream`, `batch`, `abatch`, `astream_log` calls.\n",
    "\n",
    "Output parsers accept a string or `BaseMessage` as input and can return an arbitrary type."
   ]
--- a/docs/docs/how_to/pydantic_compatibility.md
+++ b/docs/docs/how_to/pydantic_compatibility.md
@@ -1,107 +0,0 @@
-# How to use LangChain with different Pydantic versions
-
- Pydantic v2 was released in June, 2023 (https://docs.pydantic.dev/2.0/blog/pydantic-v2-final/)
- v2 contains has a number of breaking changes (https://docs.pydantic.dev/2.0/migration/)
- Pydantic v2 and v1 are under the same package name, so both versions cannot be installed at the same time
-
-## LangChain Pydantic migration plan
-
-As of `langchain>=0.0.267`, LangChain will allow users to install either Pydantic V1 or V2. 
-   * Internally LangChain will continue to [use V1](https://docs.pydantic.dev/latest/migration/#continue-using-pydantic-v1-features).
-   * During this time, users can pin their pydantic version to v1 to avoid breaking changes, or start a partial
-   migration using pydantic v2 throughout their code, but avoiding mixing v1 and v2 code for LangChain (see below).
-
-User can either pin to pydantic v1, and upgrade their code in one go once LangChain has migrated to v2 internally, or they can start a partial migration to v2, but must avoid mixing v1 and v2 code for LangChain.
-
-Below are two examples of showing how to avoid mixing pydantic v1 and v2 code in
-the case of inheritance and in the case of passing objects to LangChain.
-
-**Example 1: Extending via inheritance**
-
-**YES** 
-
-```python
-from pydantic.v1 import root_validator, validator
-from langchain_core.tools import BaseTool
-
-class CustomTool(BaseTool): # BaseTool is v1 code
-    x: int = Field(default=1)
-
-    def _run(*args, **kwargs):
-        return "hello"
-
-    @validator('x') # v1 code
-    @classmethod
-    def validate_x(cls, x: int) -> int:
-        return 1
-    
-
-CustomTool(
-    name='custom_tool',
-    description="hello",
-    x=1,
-)
-```
-
-Mixing Pydantic v2 primitives with Pydantic v1 primitives can raise cryptic errors
-
-**NO** 
-
-```python
-from pydantic import Field, field_validator # pydantic v2
-from langchain_core.tools import BaseTool
-
-class CustomTool(BaseTool): # BaseTool is v1 code
-    x: int = Field(default=1)
-
-    def _run(*args, **kwargs):
-        return "hello"
-
-    @field_validator('x') # v2 code
-    @classmethod
-    def validate_x(cls, x: int) -> int:
-        return 1
-    
-
-CustomTool( 
-    name='custom_tool',
-    description="hello",
-    x=1,
-)
-```
-
-**Example 2: Passing objects to LangChain**
-
-**YES**
-
-```python
-from langchain_core.tools import Tool
-from pydantic.v1 import BaseModel, Field # <-- Uses v1 namespace
-
-class CalculatorInput(BaseModel):
-    question: str = Field()
-
-Tool.from_function( # <-- tool uses v1 namespace
-    func=lambda question: 'hello',
-    name="Calculator",
-    description="useful for when you need to answer questions about math",
-    args_schema=CalculatorInput
-)
-```
-
-**NO**
-
-```python
-from langchain_core.tools import Tool
-from pydantic import BaseModel, Field # <-- Uses v2 namespace
-
-class CalculatorInput(BaseModel):
-    question: str = Field()
-
-Tool.from_function( # <-- tool uses v1 namespace
-    func=lambda question: 'hello',
-    name="Calculator",
-    description="useful for when you need to answer questions about math",
-    args_schema=CalculatorInput
-)
-```
--- a/docs/docs/how_to/qa_sources.ipynb
+++ b/docs/docs/how_to/qa_sources.ipynb
@@ -14,7 +14,7 @@
    "We will cover two approaches:\n",
    "\n",
    "1. Using the built-in [create_retrieval_chain](https://api.python.langchain.com/en/latest/chains/langchain.chains.retrieval.create_retrieval_chain.html), which returns sources by default;\n",
-    "2. Using a simple [LCEL](/docs/concepts#langchain-expression-language-lcel) implementation, to show the operating principle."
+    "2. Using a simple [LCEL](/docs/concepts#langchain-expression-language) implementation, to show the operating principle."
   ]
  },
  {
--- a/docs/docs/how_to/routing.ipynb
+++ b/docs/docs/how_to/routing.ipynb
@@ -323,7 +323,7 @@
   "id": "fa0f589d",
   "metadata": {},
   "source": [
-    "## Routing by semantic similarity\n",
+    "# Routing by semantic similarity\n",
    "\n",
    "One especially useful technique is to use embeddings to route a query to the most relevant prompt. Here's an example."
   ]
@@ -371,7 +371,7 @@
    "chain = (\n",
    "    {\"query\": RunnablePassthrough()}\n",
    "    | RunnableLambda(prompt_router)\n",
-    "    | ChatAnthropic(model=\"claude-3-haiku-20240307\")\n",
+    "    | ChatAnthropic(model_name=\"claude-3-haiku-20240307\")\n",
    "    | StrOutputParser()\n",
    ")"
   ]
--- a/docs/docs/how_to/semantic-chunker.ipynb
+++ b/docs/docs/how_to/semantic-chunker.ipynb
@@ -297,67 +297,13 @@
    "print(len(docs))"
   ]
  },
-  {
-   "cell_type": "markdown",
-   "source": [
-    "### Gradient\n",
-    "\n",
-    "In this method, the gradient of distance is used to split chunks along with the percentile method.\n",
-    "This method is useful when chunks are highly correlated with each other or specific to a domain e.g. legal or medical. The idea is to apply anomaly detection on gradient array so that the distribution become wider and easy to identify boundaries in highly semantic data."
-   ],
-   "metadata": {
-    "collapsed": false
-   },
-   "id": "423c6e099e94ca69"
-  },
  {
   "cell_type": "code",
   "execution_count": null,
   "id": "b1f65472",
   "metadata": {},
   "outputs": [],
-   "source": [
-    "text_splitter = SemanticChunker(\n",
-    "    OpenAIEmbeddings(), breakpoint_threshold_type=\"gradient\"\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 6,
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "Madam Speaker, Madam Vice President, our First Lady and Second Gentleman.\n"
-     ]
-    }
-   ],
-   "source": [
-    "docs = text_splitter.create_documents([state_of_the_union])\n",
-    "print(docs[0].page_content)"
-   ],
-   "metadata": {},
-   "id": "e9f393d316ce1f6c"
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 8,
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "26\n"
-     ]
-    }
-   ],
-   "source": [
-    "print(len(docs))"
-   ],
-   "metadata": {},
-   "id": "a407cd57f02a0db4"
+   "source": []
  }
 ],
 "metadata": {
--- a/docs/docs/how_to/streaming.ipynb
+++ b/docs/docs/how_to/streaming.ipynb
@@ -41,10 +41,6 @@
    "\n",
    "Let's take a look at both approaches, and try to understand how to use them.\n",
    "\n",
-    ":::info\n",
-    "For a higher-level overview of streaming techniques in LangChain, see [this section of the conceptual guide](/docs/concepts/#streaming).\n",
-    ":::\n",
-    "\n",
    "## Using Stream\n",
    "\n",
    "All `Runnable` objects implement a sync method called `stream` and an async variant called `astream`. \n",
@@ -1007,7 +1003,7 @@
   "id": "798ea891-997c-454c-bf60-43124f40ee1b",
   "metadata": {},
   "source": [
-    "Because both the model and the parser support streaming, we see streaming events from both components in real time! Kind of cool isn't it? 🦜"
+    "Because both the model and the parser support streaming, we see sreaming events from both components in real time! Kind of cool isn't it? 🦜"
   ]
  },
  {
--- a/docs/docs/how_to/structured_output.ipynb
+++ b/docs/docs/how_to/structured_output.ipynb
@@ -33,8 +33,6 @@
    "\n",
    "## The `.with_structured_output()` method\n",
    "\n",
-    "<span data-heading-keywords=\"with_structured_output\"></span>\n",
-    "\n",
    ":::info Supported models\n",
    "\n",
    "You can find a [list of models that support this method here](/docs/integrations/chat/).\n",
@@ -76,7 +74,7 @@
   "id": "a808a401-be1f-49f9-ad13-58dd68f7db5f",
   "metadata": {},
   "source": [
-    "If we want the model to return a Pydantic object, we just need to pass in the desired Pydantic class:"
+    "If we want the model to return a Pydantic object, we just need to pass in desired the Pydantic class:"
   ]
  },
  {
--- a/docs/docs/how_to/tool_calling.ipynb
+++ b/docs/docs/how_to/tool_calling.ipynb
@@ -167,83 +167,13 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 5,
+   "execution_count": 4,
   "metadata": {},
   "outputs": [],
   "source": [
    "llm_with_tools = llm.bind_tools(tools)"
   ]
  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "We can also use the `tool_choice` parameter to ensure certain behavior. For example, we can force our tool to call the multiply tool by using the following code:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 9,
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "AIMessage(content='', additional_kwargs={'tool_calls': [{'id': 'call_9cViskmLvPnHjXk9tbVla5HA', 'function': {'arguments': '{\"a\":2,\"b\":4}', 'name': 'Multiply'}, 'type': 'function'}]}, response_metadata={'token_usage': {'completion_tokens': 9, 'prompt_tokens': 103, 'total_tokens': 112}, 'model_name': 'gpt-3.5-turbo-0125', 'system_fingerprint': None, 'finish_reason': 'stop', 'logprobs': None}, id='run-095b827e-2bdd-43bb-8897-c843f4504883-0', tool_calls=[{'name': 'Multiply', 'args': {'a': 2, 'b': 4}, 'id': 'call_9cViskmLvPnHjXk9tbVla5HA'}], usage_metadata={'input_tokens': 103, 'output_tokens': 9, 'total_tokens': 112})"
-      ]
-     },
-     "execution_count": 9,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "llm_forced_to_multiply = llm.bind_tools(tools, tool_choice=\"Multiply\")\n",
-    "llm_forced_to_multiply.invoke(\"what is 2 + 4\")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "Even if we pass it something that doesn't require multiplcation - it will still call the tool!"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "We can also just force our tool to select at least one of our tools by passing in the \"any\" (or \"required\" which is OpenAI specific) keyword to the `tool_choice` parameter."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 10,
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "AIMessage(content='', additional_kwargs={'tool_calls': [{'id': 'call_mCSiJntCwHJUBfaHZVUB2D8W', 'function': {'arguments': '{\"a\":1,\"b\":2}', 'name': 'Add'}, 'type': 'function'}]}, response_metadata={'token_usage': {'completion_tokens': 15, 'prompt_tokens': 94, 'total_tokens': 109}, 'model_name': 'gpt-3.5-turbo-0125', 'system_fingerprint': None, 'finish_reason': 'stop', 'logprobs': None}, id='run-28f75260-9900-4bed-8cd3-f1579abb65e5-0', tool_calls=[{'name': 'Add', 'args': {'a': 1, 'b': 2}, 'id': 'call_mCSiJntCwHJUBfaHZVUB2D8W'}], usage_metadata={'input_tokens': 94, 'output_tokens': 15, 'total_tokens': 109})"
-      ]
-     },
-     "execution_count": 10,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "llm_forced_to_use_tool = llm.bind_tools(tools, tool_choice=\"any\")\n",
-    "llm_forced_to_use_tool.invoke(\"What day is today?\")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "As we can see, even though the prompt didn't really suggest a tool call, our LLM made one since it was forced to do so. You can look at the docs for [`bind_tool`](https://api.python.langchain.com/en/latest/chat_models/langchain_openai.chat_models.base.BaseChatOpenAI.html#langchain_openai.chat_models.base.BaseChatOpenAI.bind_tools) to learn about all the ways to customize how your LLM selects tools."
-   ]
-  },
  {
   "cell_type": "markdown",
   "metadata": {},
@@ -781,7 +711,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
-   "version": "3.11.9"
+   "version": "3.11.4"
  }
 },
 "nbformat": 4,
--- a/docs/docs/how_to/tools_builtin.ipynb
+++ b/docs/docs/how_to/tools_builtin.ipynb
@@ -36,7 +36,7 @@
    "\n",
    "When using 3rd party tools, make sure that you understand how the tool works, what permissions\n",
    "it has. Read over its documentation and check if anything is required from you\n",
-    "from a security point of view. Please see our [security](https://python.langchain.com/v0.2/docs/security/) \n",
+    "from a security point of view. Please see our [security](https://python.langchain.com/v0.1/docs/security/) \n",
    "guidelines for more information.\n",
    "\n",
    ":::\n",
--- a/docs/docs/how_to/trim_messages.ipynb
+++ b/docs/docs/how_to/trim_messages.ipynb
@@ -1,473 +0,0 @@
-{
- "cells": [
-  {
-   "cell_type": "markdown",
-   "id": "b5ee5b75-6876-4d62-9ade-5a7a808ae5a2",
-   "metadata": {},
-   "source": [
-    "# How to trim messages\n",
-    "\n",
-    "All models have finite context windows, meaning there's a limit to how many tokens they can take as input. If you have very long messages or a chain/agent that accumulates a long message is history, you'll need to manage the length of the messages you're passing in to the model.\n",
-    "\n",
-    "The `trim_messages` util provides some basic strategies for trimming a list of messages to be of a certain token length.\n",
-    "\n",
-    "## Getting the last `max_tokens` tokens\n",
-    "\n",
-    "To get the last `max_tokens` in the list of Messages we can set `strategy=\"last\"`. Notice that for our `token_counter` we can pass in a function (more on that below) or a language model (since language models have a message token counting method). It makes sense to pass in a model when you're trimming your messages to fit into the context window of that specific model:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 1,
-   "id": "c974633b-3bd0-4844-8a8f-85e3e25f13fe",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[AIMessage(content=\"Hmmm let me think.\\n\\nWhy, he's probably chasing after the last cup of coffee in the office!\"),\n",
-       " HumanMessage(content='what do you call a speechless parrot')]"
-      ]
-     },
-     "execution_count": 1,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "# pip install -U langchain-openai\n",
-    "from langchain_core.messages import (\n",
-    "    AIMessage,\n",
-    "    HumanMessage,\n",
-    "    SystemMessage,\n",
-    "    trim_messages,\n",
-    ")\n",
-    "from langchain_openai import ChatOpenAI\n",
-    "\n",
-    "messages = [\n",
-    "    SystemMessage(\"you're a good assistant, you always respond with a joke.\"),\n",
-    "    HumanMessage(\"i wonder why it's called langchain\"),\n",
-    "    AIMessage(\n",
-    "        'Well, I guess they thought \"WordRope\" and \"SentenceString\" just didn\\'t have the same ring to it!'\n",
-    "    ),\n",
-    "    HumanMessage(\"and who is harrison chasing anyways\"),\n",
-    "    AIMessage(\n",
-    "        \"Hmmm let me think.\\n\\nWhy, he's probably chasing after the last cup of coffee in the office!\"\n",
-    "    ),\n",
-    "    HumanMessage(\"what do you call a speechless parrot\"),\n",
-    "]\n",
-    "\n",
-    "trim_messages(\n",
-    "    messages,\n",
-    "    max_tokens=45,\n",
-    "    strategy=\"last\",\n",
-    "    token_counter=ChatOpenAI(model=\"gpt-4o\"),\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "d3f46654-c4b2-4136-b995-91c3febe5bf9",
-   "metadata": {},
-   "source": [
-    "If we want to always keep the initial system message we can specify `include_system=True`:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 2,
-   "id": "589b0223-3a73-44ec-8315-2dba3ee6117d",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[SystemMessage(content=\"you're a good assistant, you always respond with a joke.\"),\n",
-       " HumanMessage(content='what do you call a speechless parrot')]"
-      ]
-     },
-     "execution_count": 2,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "trim_messages(\n",
-    "    messages,\n",
-    "    max_tokens=45,\n",
-    "    strategy=\"last\",\n",
-    "    token_counter=ChatOpenAI(model=\"gpt-4o\"),\n",
-    "    include_system=True,\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "8a8b542c-04d1-4515-8d82-b999ea4fac4f",
-   "metadata": {},
-   "source": [
-    "If we want to allow splitting up the contents of a message we can specify `allow_partial=True`:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 3,
-   "id": "8c46a209-dddd-4d01-81f6-f6ae55d3225c",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[SystemMessage(content=\"you're a good assistant, you always respond with a joke.\"),\n",
-       " AIMessage(content=\"\\nWhy, he's probably chasing after the last cup of coffee in the office!\"),\n",
-       " HumanMessage(content='what do you call a speechless parrot')]"
-      ]
-     },
-     "execution_count": 3,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "trim_messages(\n",
-    "    messages,\n",
-    "    max_tokens=56,\n",
-    "    strategy=\"last\",\n",
-    "    token_counter=ChatOpenAI(model=\"gpt-4o\"),\n",
-    "    include_system=True,\n",
-    "    allow_partial=True,\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "306adf9c-41cd-495c-b4dc-e4f43dd7f8f8",
-   "metadata": {},
-   "source": [
-    "If we need to make sure that our first message (excluding the system message) is always of a specific type, we can specify `start_on`:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 4,
-   "id": "878a730b-fe44-4e9d-ab65-7b8f7b069de8",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[SystemMessage(content=\"you're a good assistant, you always respond with a joke.\"),\n",
-       " HumanMessage(content='what do you call a speechless parrot')]"
-      ]
-     },
-     "execution_count": 4,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "trim_messages(\n",
-    "    messages,\n",
-    "    max_tokens=60,\n",
-    "    strategy=\"last\",\n",
-    "    token_counter=ChatOpenAI(model=\"gpt-4o\"),\n",
-    "    include_system=True,\n",
-    "    start_on=\"human\",\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "7f5d391d-235b-4091-b2de-c22866b478f3",
-   "metadata": {},
-   "source": [
-    "## Getting the first `max_tokens` tokens\n",
-    "\n",
-    "We can perform the flipped operation of getting the *first* `max_tokens` by specifying `strategy=\"first\"`:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 5,
-   "id": "5f56ae54-1a39-4019-9351-3b494c003d5b",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[SystemMessage(content=\"you're a good assistant, you always respond with a joke.\"),\n",
-       " HumanMessage(content=\"i wonder why it's called langchain\")]"
-      ]
-     },
-     "execution_count": 5,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "trim_messages(\n",
-    "    messages,\n",
-    "    max_tokens=45,\n",
-    "    strategy=\"first\",\n",
-    "    token_counter=ChatOpenAI(model=\"gpt-4o\"),\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "ab70bf70-1e5a-4d51-b9b8-a823bf2cf532",
-   "metadata": {},
-   "source": [
-    "## Writing a custom token counter\n",
-    "\n",
-    "We can write a custom token counter function that takes in a list of messages and returns an int."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 6,
-   "id": "1c1c3b1e-2ece-49e7-a3b6-e69877c1633b",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[AIMessage(content=\"Hmmm let me think.\\n\\nWhy, he's probably chasing after the last cup of coffee in the office!\"),\n",
-       " HumanMessage(content='what do you call a speechless parrot')]"
-      ]
-     },
-     "execution_count": 6,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "from typing import List\n",
-    "\n",
-    "# pip install tiktoken\n",
-    "import tiktoken\n",
-    "from langchain_core.messages import BaseMessage, ToolMessage\n",
-    "\n",
-    "\n",
-    "def str_token_counter(text: str) -> int:\n",
-    "    enc = tiktoken.get_encoding(\"o200k_base\")\n",
-    "    return len(enc.encode(text))\n",
-    "\n",
-    "\n",
-    "def tiktoken_counter(messages: List[BaseMessage]) -> int:\n",
-    "    \"\"\"Approximately reproduce https://github.com/openai/openai-cookbook/blob/main/examples/How_to_count_tokens_with_tiktoken.ipynb\n",
-    "\n",
-    "    For simplicity only supports str Message.contents.\n",
-    "    \"\"\"\n",
-    "    num_tokens = 3  # every reply is primed with <|start|>assistant<|message|>\n",
-    "    tokens_per_message = 3\n",
-    "    tokens_per_name = 1\n",
-    "    for msg in messages:\n",
-    "        if isinstance(msg, HumanMessage):\n",
-    "            role = \"user\"\n",
-    "        elif isinstance(msg, AIMessage):\n",
-    "            role = \"assistant\"\n",
-    "        elif isinstance(msg, ToolMessage):\n",
-    "            role = \"tool\"\n",
-    "        elif isinstance(msg, SystemMessage):\n",
-    "            role = \"system\"\n",
-    "        else:\n",
-    "            raise ValueError(f\"Unsupported messages type {msg.__class__}\")\n",
-    "        num_tokens += (\n",
-    "            tokens_per_message\n",
-    "            + str_token_counter(role)\n",
-    "            + str_token_counter(msg.content)\n",
-    "        )\n",
-    "        if msg.name:\n",
-    "            num_tokens += tokens_per_name + str_token_counter(msg.name)\n",
-    "    return num_tokens\n",
-    "\n",
-    "\n",
-    "trim_messages(\n",
-    "    messages,\n",
-    "    max_tokens=45,\n",
-    "    strategy=\"last\",\n",
-    "    token_counter=tiktoken_counter,\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "4b2a672b-c007-47c5-9105-617944dc0a6a",
-   "metadata": {},
-   "source": [
-    "## Chaining\n",
-    "\n",
-    "`trim_messages` can be used in an imperatively (like above) or declaratively, making it easy to compose with other components in a chain"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 7,
-   "id": "96aa29b2-01e0-437c-a1ab-02fb0141cb57",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "AIMessage(content='A \"polygon\"! Because it\\'s a \"poly-gone\" silent!', response_metadata={'token_usage': {'completion_tokens': 14, 'prompt_tokens': 32, 'total_tokens': 46}, 'model_name': 'gpt-4o-2024-05-13', 'system_fingerprint': 'fp_319be4768e', 'finish_reason': 'stop', 'logprobs': None}, id='run-64cc4575-14d1-4f3f-b4af-97f24758f703-0', usage_metadata={'input_tokens': 32, 'output_tokens': 14, 'total_tokens': 46})"
-      ]
-     },
-     "execution_count": 7,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "llm = ChatOpenAI(model=\"gpt-4o\")\n",
-    "\n",
-    "# Notice we don't pass in messages. This creates\n",
-    "# a RunnableLambda that takes messages as input\n",
-    "trimmer = trim_messages(\n",
-    "    max_tokens=45,\n",
-    "    strategy=\"last\",\n",
-    "    token_counter=llm,\n",
-    "    include_system=True,\n",
-    ")\n",
-    "\n",
-    "chain = trimmer | llm\n",
-    "chain.invoke(messages)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "4d91d390-e7f7-467b-ad87-d100411d7a21",
-   "metadata": {},
-   "source": [
-    "Looking at the LangSmith trace we can see that before the messages are passed to the model they are first trimmed: https://smith.langchain.com/public/65af12c4-c24d-4824-90f0-6547566e59bb/r\n",
-    "\n",
-    "Looking at just the trimmer, we can see that it's a Runnable object that can be invoked like all Runnables:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 8,
-   "id": "1ff02d0a-353d-4fac-a77c-7c2c5262abd9",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[SystemMessage(content=\"you're a good assistant, you always respond with a joke.\"),\n",
-       " HumanMessage(content='what do you call a speechless parrot')]"
-      ]
-     },
-     "execution_count": 8,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "trimmer.invoke(messages)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "dc4720c8-4062-4ebc-9385-58411202ce6e",
-   "metadata": {},
-   "source": [
-    "## Using with ChatMessageHistory\n",
-    "\n",
-    "Trimming messages is especially useful when [working with chat histories](/docs/how_to/message_history/), which can get arbitrarily long:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 14,
-   "id": "a9517858-fc2f-4dc3-898d-bf98a0e905a0",
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stderr",
-     "output_type": "stream",
-     "text": [
-      "Parent run c87e2f1b-81ad-4fa7-bfd9-ce6edb29a482 not found for run 7892ee8f-0669-4d6b-a2ca-ef8aae81042a. Treating as a root run.\n"
-     ]
-    },
-    {
-     "data": {
-      "text/plain": [
-       "AIMessage(content=\"A polygon! Because it's a parrot gone quiet!\", response_metadata={'token_usage': {'completion_tokens': 11, 'prompt_tokens': 32, 'total_tokens': 43}, 'model_name': 'gpt-4o-2024-05-13', 'system_fingerprint': 'fp_319be4768e', 'finish_reason': 'stop', 'logprobs': None}, id='run-72dad96e-8b58-45f4-8c08-21f9f1a6b68f-0', usage_metadata={'input_tokens': 32, 'output_tokens': 11, 'total_tokens': 43})"
-      ]
-     },
-     "execution_count": 14,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "from langchain_core.chat_history import InMemoryChatMessageHistory\n",
-    "from langchain_core.runnables.history import RunnableWithMessageHistory\n",
-    "\n",
-    "chat_history = InMemoryChatMessageHistory(messages=messages[:-1])\n",
-    "\n",
-    "\n",
-    "def dummy_get_session_history(session_id):\n",
-    "    if session_id != \"1\":\n",
-    "        raise InMemoryChatMessageHistory()\n",
-    "    return chat_history\n",
-    "\n",
-    "\n",
-    "llm = ChatOpenAI(model=\"gpt-4o\")\n",
-    "\n",
-    "trimmer = trim_messages(\n",
-    "    max_tokens=45,\n",
-    "    strategy=\"last\",\n",
-    "    token_counter=llm,\n",
-    "    include_system=True,\n",
-    ")\n",
-    "\n",
-    "chain = trimmer | llm\n",
-    "chain_with_history = RunnableWithMessageHistory(chain, dummy_get_session_history)\n",
-    "chain_with_history.invoke(\n",
-    "    [HumanMessage(\"what do you call a speechless parrot\")],\n",
-    "    config={\"configurable\": {\"session_id\": \"1\"}},\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "556b7b4c-43cb-41de-94fc-1a41f4ec4d2e",
-   "metadata": {},
-   "source": [
-    "Looking at the LangSmith trace we can see that we retrieve all of our messages but before the messages are passed to the model they are trimmed to be just the system message and last human message: https://smith.langchain.com/public/17dd700b-9994-44ca-930c-116e00997315/r"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "75dc7b84-b92f-44e7-8beb-ba22398e4efb",
-   "metadata": {},
-   "source": [
-    "## API reference\n",
-    "\n",
-    "For a complete description of all arguments head to the API reference: https://api.python.langchain.com/en/latest/messages/langchain_core.messages.utils.trim_messages.html"
-   ]
-  }
- ],
- "metadata": {
-  "kernelspec": {
-   "display_name": "poetry-venv-2",
-   "language": "python",
-   "name": "poetry-venv-2"
-  },
-  "language_info": {
-   "codemirror_mode": {
-    "name": "ipython",
-    "version": 3
-   },
-   "file_extension": ".py",
-   "mimetype": "text/x-python",
-   "name": "python",
-   "nbconvert_exporter": "python",
-   "pygments_lexer": "ipython3",
-   "version": "3.9.1"
-  }
- },
- "nbformat": 4,
- "nbformat_minor": 5
-}
--- a/docs/docs/integrations/callbacks/upstash_ratelimit.ipynb
+++ b/docs/docs/integrations/callbacks/upstash_ratelimit.ipynb
@@ -1,245 +0,0 @@
-{
- "cells": [
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "# Upstash Ratelimit Callback\n",
-    "\n",
-    "In this guide, we will go over how to add rate limiting based on number of requests or the number of tokens using `UpstashRatelimitHandler`. This handler uses [ratelimit library of Upstash](https://github.com/upstash/ratelimit-py/), which utilizes [Upstash Redis](https://upstash.com/docs/redis/overall/getstarted).\n",
-    "\n",
-    "Upstash Ratelimit works by sending an HTTP request to Upstash Redis everytime the `limit` method is called. Remaining tokens/requests of the user are checked and updated. Based on the remaining tokens, we can stop the execution of costly operations like invoking an LLM or querying a vector store:\n",
-    "\n",
-    "```py\n",
-    "response = ratelimit.limit()\n",
-    "if response.allowed:\n",
-    "    execute_costly_operation()\n",
-    "```\n",
-    "\n",
-    "`UpstashRatelimitHandler` allows you to incorporate the ratelimit logic into your chain in a few minutes.\n",
-    "\n",
-    "First, you will need to go to [the Upstash Console](https://console.upstash.com/login) and create a redis database ([see our docs](https://upstash.com/docs/redis/overall/getstarted)). After creating a database, you will need to set the environment variables:\n",
-    "\n",
-    "```\n",
-    "UPSTASH_REDIS_REST_URL=\"****\"\n",
-    "UPSTASH_REDIS_REST_TOKEN=\"****\"\n",
-    "```\n",
-    "\n",
-    "Next, you will need to install Upstash Ratelimit and Redis library with:\n",
-    "\n",
-    "```\n",
-    "pip install upstash-ratelimit upstash-redis\n",
-    "```\n",
-    "\n",
-    "You are now ready to add rate limiting to your chain!\n",
-    "\n",
-    "## Ratelimiting Per Request\n",
-    "\n",
-    "Let's imagine that we want to allow our users to invoke our chain 10 times per minute. Achieving this is as simple as:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 21,
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stderr",
-     "output_type": "stream",
-     "text": [
-      "Error in UpstashRatelimitHandler.on_chain_start callback: UpstashRatelimitError('Request limit reached!')\n"
-     ]
-    },
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "Handling ratelimit. <class 'langchain_community.callbacks.upstash_ratelimit_callback.UpstashRatelimitError'>\n"
-     ]
-    }
-   ],
-   "source": [
-    "# set env variables\n",
-    "import os\n",
-    "\n",
-    "os.environ[\"UPSTASH_REDIS_REST_URL\"] = \"****\"\n",
-    "os.environ[\"UPSTASH_REDIS_REST_TOKEN\"] = \"****\"\n",
-    "\n",
-    "from langchain_community.callbacks import UpstashRatelimitError, UpstashRatelimitHandler\n",
-    "from langchain_core.runnables import RunnableLambda\n",
-    "from upstash_ratelimit import FixedWindow, Ratelimit\n",
-    "from upstash_redis import Redis\n",
-    "\n",
-    "# create ratelimit\n",
-    "ratelimit = Ratelimit(\n",
-    "    redis=Redis.from_env(),\n",
-    "    # 10 requests per window, where window size is 60 seconds:\n",
-    "    limiter=FixedWindow(max_requests=10, window=60),\n",
-    ")\n",
-    "\n",
-    "# create handler\n",
-    "user_id = \"user_id\"  # should be a method which gets the user id\n",
-    "handler = UpstashRatelimitHandler(identifier=user_id, request_ratelimit=ratelimit)\n",
-    "\n",
-    "# create mock chain\n",
-    "chain = RunnableLambda(str)\n",
-    "\n",
-    "# invoke chain with handler:\n",
-    "try:\n",
-    "    result = chain.invoke(\"Hello world!\", config={\"callbacks\": [handler]})\n",
-    "except UpstashRatelimitError:\n",
-    "    print(\"Handling ratelimit.\", UpstashRatelimitError)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "Note that we pass the handler to the `invoke` method instead of passing the handler when defining the chain.\n",
-    "\n",
-    "For rate limiting algorithms other than `FixedWindow`, see [upstash-ratelimit docs](https://github.com/upstash/ratelimit-py?tab=readme-ov-file#ratelimiting-algorithms).\n",
-    "\n",
-    "Before executing any steps in our pipeline, ratelimit will check whether the user has passed the request limit. If so, `UpstashRatelimitError` is raised.\n",
-    "\n",
-    "## Ratelimiting Per Token\n",
-    "\n",
-    "Another option is to rate limit chain invokations based on:\n",
-    "1. number of tokens in prompt\n",
-    "2. number of tokens in prompt and LLM completion\n",
-    "\n",
-    "This only works if you have an LLM in your chain. Another requirement is that the LLM you are using should return the token usage in it's `LLMOutput`.\n",
-    "\n",
-    "### How it works\n",
-    "\n",
-    "The handler will get the remaining tokens before calling the LLM. If the remaining tokens is more than 0, LLM will be called. Otherwise `UpstashRatelimitError` will be raised.\n",
-    "\n",
-    "After LLM is called, token usage information will be used to subtracted from the remaining tokens of the user. No error is raised at this stage of the chain.\n",
-    "\n",
-    "### Configuration\n",
-    "\n",
-    "For the first configuration, simply initialize the handler like this:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "ratelimit = Ratelimit(\n",
-    "    redis=Redis.from_env(),\n",
-    "    # 1000 tokens per window, where window size is 60 seconds:\n",
-    "    limiter=FixedWindow(max_requests=1000, window=60),\n",
-    ")\n",
-    "\n",
-    "handler = UpstashRatelimitHandler(identifier=user_id, token_ratelimit=ratelimit)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "For the second configuration, here is how to initialize the handler:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "ratelimit = Ratelimit(\n",
-    "    redis=Redis.from_env(),\n",
-    "    # 1000 tokens per window, where window size is 60 seconds:\n",
-    "    limiter=FixedWindow(max_requests=1000, window=60),\n",
-    ")\n",
-    "\n",
-    "handler = UpstashRatelimitHandler(\n",
-    "    identifier=user_id,\n",
-    "    token_ratelimit=ratelimit,\n",
-    "    include_output_tokens=True,  # set to True\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "You can also employ ratelimiting based on requests and tokens at the same time, simply by passing both `request_ratelimit` and `token_ratelimit` parameters.\n",
-    "\n",
-    "Here is an example with a chain utilizing an LLM:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 22,
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stderr",
-     "output_type": "stream",
-     "text": [
-      "Error in UpstashRatelimitHandler.on_llm_start callback: UpstashRatelimitError('Token limit reached!')\n"
-     ]
-    },
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "Handling ratelimit. <class 'langchain_community.callbacks.upstash_ratelimit_callback.UpstashRatelimitError'>\n"
-     ]
-    }
-   ],
-   "source": [
-    "# set env variables\n",
-    "import os\n",
-    "\n",
-    "os.environ[\"UPSTASH_REDIS_REST_URL\"] = \"****\"\n",
-    "os.environ[\"UPSTASH_REDIS_REST_TOKEN\"] = \"****\"\n",
-    "os.environ[\"OPENAI_API_KEY\"] = \"****\"\n",
-    "\n",
-    "from langchain_community.callbacks import UpstashRatelimitError, UpstashRatelimitHandler\n",
-    "from langchain_core.runnables import RunnableLambda\n",
-    "from langchain_openai import ChatOpenAI\n",
-    "from upstash_ratelimit import FixedWindow, Ratelimit\n",
-    "from upstash_redis import Redis\n",
-    "\n",
-    "# create ratelimit\n",
-    "ratelimit = Ratelimit(\n",
-    "    redis=Redis.from_env(),\n",
-    "    # 500 tokens per window, where window size is 60 seconds:\n",
-    "    limiter=FixedWindow(max_requests=500, window=60),\n",
-    ")\n",
-    "\n",
-    "# create handler\n",
-    "user_id = \"user_id\"  # should be a method which gets the user id\n",
-    "handler = UpstashRatelimitHandler(identifier=user_id, token_ratelimit=ratelimit)\n",
-    "\n",
-    "# create mock chain\n",
-    "as_str = RunnableLambda(str)\n",
-    "model = ChatOpenAI()\n",
-    "\n",
-    "chain = as_str | model\n",
-    "\n",
-    "# invoke chain with handler:\n",
-    "try:\n",
-    "    result = chain.invoke(\"Hello world!\", config={\"callbacks\": [handler]})\n",
-    "except UpstashRatelimitError:\n",
-    "    print(\"Handling ratelimit.\", UpstashRatelimitError)"
-   ]
-  }
- ],
- "metadata": {
-  "kernelspec": {
-   "display_name": "lc39",
-   "language": "python",
-   "name": "python3"
-  },
-  "language_info": {
-   "name": "python",
-   "version": "3.9.19"
-  }
- },
- "nbformat": 4,
- "nbformat_minor": 2
-}
--- a/docs/docs/integrations/chat/azure_chat_openai.ipynb
+++ b/docs/docs/integrations/chat/azure_chat_openai.ipynb
@@ -23,11 +23,13 @@
   ]
  },
  {
-   "cell_type": "code",
-   "execution_count": null,
+   "cell_type": "raw",
   "id": "d83ba7de",
-   "metadata": {},
-   "outputs": [],
+   "metadata": {
+    "vscode": {
+     "languageId": "raw"
+    }
+   },
   "source": [
    "%pip install -qU langchain-openai"
   ]
--- a/docs/docs/integrations/chat/cohere.ipynb
+++ b/docs/docs/integrations/chat/cohere.ipynb
@@ -201,7 +201,7 @@
   "source": [
    "## Chaining\n",
    "\n",
-    "You can also easily combine with a prompt template for easy structuring of user input. We can do this using [LCEL](/docs/concepts#langchain-expression-language-lcel)"
+    "You can also easily combine with a prompt template for easy structuring of user input. We can do this using [LCEL](/docs/concepts#langchain-expression-language)"
   ]
  },
  {
--- a/docs/docs/integrations/chat/deepinfra.ipynb
+++ b/docs/docs/integrations/chat/deepinfra.ipynb
@@ -98,78 +98,6 @@
    ")\n",
    "chat.invoke(messages)"
   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "466c3cb41ace1410",
-   "metadata": {},
-   "source": [
-    "# Tool Calling\n",
-    "\n",
-    "DeepInfra currently supports only invoke and async invoke tool calling.\n",
-    "\n",
-    "For a complete list of models that support tool calling, please refer to our [tool calling documentation](https://deepinfra.com/docs/advanced/function_calling)."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "ddc4f4299763651c",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "import asyncio\n",
-    "\n",
-    "from dotenv import find_dotenv, load_dotenv\n",
-    "from langchain_community.chat_models import ChatDeepInfra\n",
-    "from langchain_core.messages import HumanMessage\n",
-    "from langchain_core.pydantic_v1 import BaseModel\n",
-    "from langchain_core.tools import tool\n",
-    "\n",
-    "model_name = \"meta-llama/Meta-Llama-3-70B-Instruct\"\n",
-    "\n",
-    "_ = load_dotenv(find_dotenv())\n",
-    "\n",
-    "\n",
-    "# Langchain tool\n",
-    "@tool\n",
-    "def foo(something):\n",
-    "    \"\"\"\n",
-    "    Called when foo\n",
-    "    \"\"\"\n",
-    "    pass\n",
-    "\n",
-    "\n",
-    "# Pydantic class\n",
-    "class Bar(BaseModel):\n",
-    "    \"\"\"\n",
-    "    Called when Bar\n",
-    "    \"\"\"\n",
-    "\n",
-    "    pass\n",
-    "\n",
-    "\n",
-    "llm = ChatDeepInfra(model=model_name)\n",
-    "tools = [foo, Bar]\n",
-    "llm_with_tools = llm.bind_tools(tools)\n",
-    "messages = [\n",
-    "    HumanMessage(\"Foo and bar, please.\"),\n",
-    "]\n",
-    "\n",
-    "response = llm_with_tools.invoke(messages)\n",
-    "print(response.tool_calls)\n",
-    "# [{'name': 'foo', 'args': {'something': None}, 'id': 'call_Mi4N4wAtW89OlbizFE1aDxDj'}, {'name': 'Bar', 'args': {}, 'id': 'call_daiE0mW454j2O1KVbmET4s2r'}]\n",
-    "\n",
-    "\n",
-    "async def call_ainvoke():\n",
-    "    result = await llm_with_tools.ainvoke(messages)\n",
-    "    print(result.tool_calls)\n",
-    "\n",
-    "\n",
-    "# Async call\n",
-    "asyncio.run(call_ainvoke())\n",
-    "# [{'name': 'foo', 'args': {'something': None}, 'id': 'call_ZH7FetmgSot4LHcMU6CEb8tI'}, {'name': 'Bar', 'args': {}, 'id': 'call_2MQhDifAJVoijZEvH8PeFSVB'}]"
-   ]
  }
 ],
 "metadata": {
--- a/docs/docs/integrations/chat/fireworks.ipynb
+++ b/docs/docs/integrations/chat/fireworks.ipynb
@@ -147,7 +147,7 @@
   "source": [
    "# Tool Calling\n",
    "\n",
-    "Fireworks offers the `FireFunction-v2` tool calling model. You can use it for structured output and function calling use cases:"
+    "Fireworks offers the [`FireFunction-v1` tool calling model](https://fireworks.ai/blog/firefunction-v1-gpt-4-level-function-calling). You can use it for structured output and function calling use cases:"
   ]
  },
  {
@@ -180,7 +180,7 @@
    "\n",
    "\n",
    "chat = ChatFireworks(\n",
-    "    model=\"accounts/fireworks/models/firefunction-v2\",\n",
+    "    model=\"accounts/fireworks/models/firefunction-v1\",\n",
    ").bind_tools([ExtractFields])\n",
    "\n",
    "result = chat.invoke(\"I am a 27 year old named Erick\")\n",
--- a/docs/docs/integrations/chat/google_vertex_ai_palm.ipynb
+++ b/docs/docs/integrations/chat/google_vertex_ai_palm.ipynb
@@ -2,50 +2,33 @@
 "cells": [
  {
   "cell_type": "raw",
-   "id": "afaf8039",
   "metadata": {},
   "source": [
    "---\n",
    "sidebar_label: Google Cloud Vertex AI\n",
+    "keywords: [gemini, vertex, ChatVertexAI, gemini-pro]\n",
    "---"
   ]
  },
  {
   "cell_type": "markdown",
-   "id": "e49f1e0d",
   "metadata": {},
   "source": [
    "# ChatVertexAI\n",
    "\n",
-    "This page provides a quick overview for getting started with VertexAI [chat models](/docs/concepts/#chat-models). For detailed documentation of all ChatVertexAI features and configurations head to the [API reference](https://api.python.langchain.com/en/latest/chat_models/langchain_google_vertexai.chat_models.ChatVertexAI.html).\n",
+    "Note: This is separate from the Google PaLM integration. Google has chosen to offer an enterprise version of PaLM through GCP, and this supports the models made available through there. \n",
    "\n",
-    "ChatVertexAI exposes all foundational models available in Google Cloud, like `gemini-1.5-pro`, `gemini-1.5-flash`, etc. For a full and updated list of available models visit [VertexAI documentation](https://cloud.google.com/vertex-ai/docs/generative-ai/model-reference/overview).\n",
+    "ChatVertexAI exposes all foundational models available in Google Cloud:\n",
    "\n",
-    ":::info Google Cloud VertexAI vs Google PaLM\n",
+    "- Gemini (`gemini-pro` and `gemini-pro-vision`)\n",
+    "- PaLM 2 for Text (`text-bison`)\n",
+    "- Codey for Code Generation (`codechat-bison`)\n",
    "\n",
-    "The Google Cloud VertexAI integration is separate from the [Google PaLM integration](/docs/integrations/chat/google_generative_ai/). Google has chosen to offer an enterprise version of PaLM through GCP, and this supports the models made available through there. \n",
+    "For a full and updated list of available models visit [VertexAI documentation](https://cloud.google.com/vertex-ai/docs/generative-ai/model-reference/overview).\n",
    "\n",
-    ":::\n",
+    "By default, Google Cloud [does not use](https://cloud.google.com/vertex-ai/docs/generative-ai/data-governance#foundation_model_development) customer data to train its foundation models as part of Google Cloud`s AI/ML Privacy Commitment. More details about how Google processes data can also be found in [Google's Customer Data Processing Addendum (CDPA)](https://cloud.google.com/terms/data-processing-addendum).\n",
    "\n",
-    "## Overview\n",
-    "### Integration details\n",
-    "\n",
-    "| Class | Package | Local | Serializable | [JS support](https://js.langchain.com/v0.2/docs/integrations/chat/google_vertex_ai) | Package downloads | Package latest |\n",
-    "| :--- | :--- | :---: | :---: |  :---: | :---: | :---: |\n",
-    "| [ChatVertexAI](https://api.python.langchain.com/en/latest/chat_models/langchain_google_vertexai.chat_models.ChatVertexAI.html) | [langchain-google-vertexai](https://api.python.langchain.com/en/latest/google_vertexai_api_reference.html) | ❌ | beta | ✅ | ![PyPI - Downloads](https://img.shields.io/pypi/dm/langchain-google-vertexai?style=flat-square&label=%20) | ![PyPI - Version](https://img.shields.io/pypi/v/langchain-google-vertexai?style=flat-square&label=%20) |\n",
-    "\n",
-    "### Model features\n",
-    "| [Tool calling](/docs/how_to/tool_calling/) | [Structured output](/docs/how_to/structured_output/) | JSON mode | [Image input](/docs/how_to/multimodal_inputs/) | Audio input | Video input | [Token-level streaming](/docs/how_to/chat_streaming/) | Native async | [Token usage](/docs/how_to/chat_token_usage_tracking/) | [Logprobs](/docs/how_to/logprobs/) |\n",
-    "| :---: | :---: | :---: | :---: |  :---: | :---: | :---: | :---: | :---: | :---: |\n",
-    "| ✅ | ✅ | ❌ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | \n",
-    "\n",
-    "## Setup\n",
-    "\n",
-    "To access VertexAI models you'll need to create a Google Cloud Platform account, set up credentials, and install the `langchain-google-vertexai` integration package.\n",
-    "\n",
-    "### Credentials\n",
-    "\n",
-    "To use the integration you must:\n",
+    "To use `Google Cloud Vertex AI` PaLM you must have the `langchain-google-vertexai` Python package installed and either:\n",
    "- Have credentials configured for your environment (gcloud, workload identity, etc...)\n",
    "- Store the path to a service account JSON file as the GOOGLE_APPLICATION_CREDENTIALS environment variable\n",
    "\n",
@@ -54,156 +37,432 @@
    "For more information, see: \n",
    "- https://cloud.google.com/docs/authentication/application-default-credentials#GAC\n",
    "- https://googleapis.dev/python/google-auth/latest/reference/google.auth.html#module-google.auth\n",
-    "\n",
-    "If you want to get automated tracing of your model calls you can also set your [LangSmith](https://docs.smith.langchain.com/) API key by uncommenting below:"
+    "\n"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "%pip install --upgrade --quiet  langchain-google-vertexai"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 1,
-   "id": "a15d341e-3e26-4ca3-830b-5aab30ed66de",
   "metadata": {},
   "outputs": [],
   "source": [
-    "# os.environ[\"LANGSMITH_API_KEY\"] = getpass.getpass(\"Enter your LangSmith API key: \")\n",
-    "# os.environ[\"LANGSMITH_TRACING\"] = \"true\""
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "0730d6a1-c893-4840-9817-5e5251676d5d",
-   "metadata": {},
-   "source": [
-    "### Installation\n",
-    "\n",
-    "The LangChain VertexAI integration lives in the `langchain-google-vertexai` package:"
+    "from langchain_core.prompts import ChatPromptTemplate\n",
+    "from langchain_google_vertexai import ChatVertexAI"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 2,
-   "id": "652d6238-1f87-422a-b135-f5abbb8652fc",
+   "execution_count": null,
   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "Note: you may need to restart the kernel to use updated packages.\n"
-     ]
-    }
-   ],
-   "source": [
-    "%pip install -qU langchain-google-vertexai"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "a38cde65-254d-4219-a441-068766c0d4b5",
-   "metadata": {},
-   "source": [
-    "## Instantiation\n",
-    "\n",
-    "Now we can instantiate our model object and generate chat completions:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 3,
-   "id": "cb09c344-1836-4e0c-acf8-11d13ac1dbae",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "from langchain_google_vertexai import ChatVertexAI\n",
-    "\n",
-    "llm = ChatVertexAI(\n",
-    "    model=\"gemini-1.5-flash-001\",\n",
-    "    temperature=0,\n",
-    "    max_tokens=None,\n",
-    "    max_retries=6,\n",
-    "    stop=None,\n",
-    "    # other params...\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "2b4f3e15",
-   "metadata": {},
-   "source": [
-    "## Invocation"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 4,
-   "id": "62e0dbc3",
-   "metadata": {
-    "tags": []
-   },
   "outputs": [
    {
     "data": {
      "text/plain": [
-       "AIMessage(content=\"J'adore programmer. \\n\", response_metadata={'is_blocked': False, 'safety_ratings': [{'category': 'HARM_CATEGORY_HATE_SPEECH', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_DANGEROUS_CONTENT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_HARASSMENT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_SEXUALLY_EXPLICIT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}], 'usage_metadata': {'prompt_token_count': 20, 'candidates_token_count': 7, 'total_token_count': 27}}, id='run-7032733c-d05c-4f0c-a17a-6c575fdd1ae0-0', usage_metadata={'input_tokens': 20, 'output_tokens': 7, 'total_tokens': 27})"
+       "AIMessage(content=\" J'aime la programmation.\")"
      ]
     },
-     "execution_count": 4,
+     "execution_count": null,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
-    "messages = [\n",
-    "    (\n",
-    "        \"system\",\n",
-    "        \"You are a helpful assistant that translates English to French. Translate the user sentence.\",\n",
-    "    ),\n",
-    "    (\"human\", \"I love programming.\"),\n",
-    "]\n",
-    "ai_msg = llm.invoke(messages)\n",
-    "ai_msg"
+    "system = \"You are a helpful assistant who translate English to French\"\n",
+    "human = \"Translate this sentence from English to French. I love programming.\"\n",
+    "prompt = ChatPromptTemplate.from_messages([(\"system\", system), (\"human\", human)])\n",
+    "\n",
+    "chat = ChatVertexAI()\n",
+    "\n",
+    "chain = prompt | chat\n",
+    "chain.invoke({})"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "Gemini doesn't support SystemMessage at the moment, but it can be added to the first human message in the row. If you want such behavior, just set the `convert_system_message_to_human` to `True`:"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 5,
-   "id": "d86145b3-bfef-46e8-b227-4dda5c9c2705",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [
+    {
+     "data": {
+      "text/plain": [
+       "AIMessage(content=\"J'aime la programmation.\")"
+      ]
+     },
+     "execution_count": null,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "system = \"You are a helpful assistant who translate English to French\"\n",
+    "human = \"Translate this sentence from English to French. I love programming.\"\n",
+    "prompt = ChatPromptTemplate.from_messages([(\"system\", system), (\"human\", human)])\n",
+    "\n",
+    "chat = ChatVertexAI(model=\"gemini-pro\", convert_system_message_to_human=True)\n",
+    "\n",
+    "chain = prompt | chat\n",
+    "chain.invoke({})"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "If we want to construct a simple chain that takes user specified parameters:"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [
+    {
+     "data": {
+      "text/plain": [
+       "AIMessage(content=' プログラミングが大好きです')"
+      ]
+     },
+     "execution_count": null,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "system = (\n",
+    "    \"You are a helpful assistant that translates {input_language} to {output_language}.\"\n",
+    ")\n",
+    "human = \"{text}\"\n",
+    "prompt = ChatPromptTemplate.from_messages([(\"system\", system), (\"human\", human)])\n",
+    "\n",
+    "chat = ChatVertexAI()\n",
+    "\n",
+    "chain = prompt | chat\n",
+    "\n",
+    "chain.invoke(\n",
+    "    {\n",
+    "        \"input_language\": \"English\",\n",
+    "        \"output_language\": \"Japanese\",\n",
+    "        \"text\": \"I love programming\",\n",
+    "    }\n",
+    ")"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "## Code generation chat models\n",
+    "You can now leverage the Codey API for code chat within Vertex AI. The model available is:\n",
+    "- `codechat-bison`: for code assistance"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
-      "J'adore programmer. \n",
-      "\n"
+      " ```python\n",
+      "def is_prime(n):\n",
+      "  \"\"\"\n",
+      "  Check if a number is prime.\n",
+      "\n",
+      "  Args:\n",
+      "    n: The number to check.\n",
+      "\n",
+      "  Returns:\n",
+      "    True if n is prime, False otherwise.\n",
+      "  \"\"\"\n",
+      "\n",
+      "  # If n is 1, it is not prime.\n",
+      "  if n == 1:\n",
+      "    return False\n",
+      "\n",
+      "  # Iterate over all numbers from 2 to the square root of n.\n",
+      "  for i in range(2, int(n ** 0.5) + 1):\n",
+      "    # If n is divisible by any number from 2 to its square root, it is not prime.\n",
+      "    if n % i == 0:\n",
+      "      return False\n",
+      "\n",
+      "  # If n is divisible by no number from 2 to its square root, it is prime.\n",
+      "  return True\n",
+      "\n",
+      "\n",
+      "def find_prime_numbers(n):\n",
+      "  \"\"\"\n",
+      "  Find all prime numbers up to a given number.\n",
+      "\n",
+      "  Args:\n",
+      "    n: The upper bound for the prime numbers to find.\n",
+      "\n",
+      "  Returns:\n",
+      "    A list of all prime numbers up to n.\n",
+      "  \"\"\"\n",
+      "\n",
+      "  # Create a list of all numbers from 2 to n.\n",
+      "  numbers = list(range(2, n + 1))\n",
+      "\n",
+      "  # Iterate over the list of numbers and remove any that are not prime.\n",
+      "  for number in numbers:\n",
+      "    if not is_prime(number):\n",
+      "      numbers.remove(number)\n",
+      "\n",
+      "  # Return the list of prime numbers.\n",
+      "  return numbers\n",
+      "```\n"
     ]
    }
   ],
   "source": [
-    "print(ai_msg.content)"
+    "chat = ChatVertexAI(model=\"codechat-bison\", max_tokens=1000, temperature=0.5)\n",
+    "\n",
+    "message = chat.invoke(\"Write a Python function generating all prime numbers\")\n",
+    "print(message.content)"
   ]
  },
  {
   "cell_type": "markdown",
-   "id": "18e2bfc0-7e78-4528-a73f-499ac150dca8",
   "metadata": {},
   "source": [
-    "## Chaining\n",
+    "## Full generation info\n",
    "\n",
-    "We can [chain](/docs/how_to/sequence/) our model with a prompt template like so:"
+    "We can use the `generate` method to get back extra metadata like [safety attributes](https://cloud.google.com/vertex-ai/docs/generative-ai/learn/responsible-ai#safety_attribute_confidence_scoring) and not just chat completions\n",
+    "\n",
+    "Note that the `generation_info` will be different depending if you're using a gemini model or not.\n",
+    "\n",
+    "### Gemini model\n",
+    "\n",
+    "`generation_info` will include:\n",
+    "\n",
+    "- `is_blocked`: whether generation was blocked or not\n",
+    "- `safety_ratings`: safety ratings' categories and probability labels"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 4,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "from pprint import pprint\n",
+    "\n",
+    "from langchain_core.messages import HumanMessage\n",
+    "from langchain_google_vertexai import HarmBlockThreshold, HarmCategory"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 5,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "{'citation_metadata': None,\n",
+      " 'is_blocked': False,\n",
+      " 'safety_ratings': [{'blocked': False,\n",
+      "                     'category': 'HARM_CATEGORY_HATE_SPEECH',\n",
+      "                     'probability_label': 'NEGLIGIBLE'},\n",
+      "                    {'blocked': False,\n",
+      "                     'category': 'HARM_CATEGORY_DANGEROUS_CONTENT',\n",
+      "                     'probability_label': 'NEGLIGIBLE'},\n",
+      "                    {'blocked': False,\n",
+      "                     'category': 'HARM_CATEGORY_HARASSMENT',\n",
+      "                     'probability_label': 'NEGLIGIBLE'},\n",
+      "                    {'blocked': False,\n",
+      "                     'category': 'HARM_CATEGORY_SEXUALLY_EXPLICIT',\n",
+      "                     'probability_label': 'NEGLIGIBLE'}],\n",
+      " 'usage_metadata': {'candidates_token_count': 6,\n",
+      "                    'prompt_token_count': 12,\n",
+      "                    'total_token_count': 18}}\n"
+     ]
+    }
+   ],
+   "source": [
+    "human = \"Translate this sentence from English to French. I love programming.\"\n",
+    "messages = [HumanMessage(content=human)]\n",
+    "\n",
+    "\n",
+    "chat = ChatVertexAI(\n",
+    "    model_name=\"gemini-pro\",\n",
+    "    safety_settings={\n",
+    "        HarmCategory.HARM_CATEGORY_HATE_SPEECH: HarmBlockThreshold.BLOCK_LOW_AND_ABOVE\n",
+    "    },\n",
+    ")\n",
+    "\n",
+    "result = chat.generate([messages])\n",
+    "pprint(result.generations[0][0].generation_info)"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "### Non-gemini model\n",
+    "\n",
+    "`generation_info` will include:\n",
+    "\n",
+    "- `is_blocked`: whether generation was blocked or not\n",
+    "- `safety_attributes`: a dictionary mapping safety attributes to their scores"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 6,
-   "id": "e197d1d7-a070-4c96-9f8a-a0e86d046e0b",
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      "{'errors': (),\n",
+      " 'grounding_metadata': {'citations': [], 'search_queries': []},\n",
+      " 'is_blocked': False,\n",
+      " 'safety_attributes': [{'Derogatory': 0.1, 'Insult': 0.1, 'Sexual': 0.2}],\n",
+      " 'usage_metadata': {'candidates_billable_characters': 88.0,\n",
+      "                    'candidates_token_count': 24.0,\n",
+      "                    'prompt_billable_characters': 58.0,\n",
+      "                    'prompt_token_count': 12.0}}\n"
+     ]
+    }
+   ],
+   "source": [
+    "chat = ChatVertexAI()  # default is `chat-bison`\n",
+    "\n",
+    "result = chat.generate([messages])\n",
+    "pprint(result.generations[0][0].generation_info)"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "## Tool calling (a.k.a. function calling) with Gemini\n",
+    "\n",
+    "We can pass tool definitions to Gemini models to get the model to invoke those tools when appropriate. This is useful not only for LLM-powered tool use but also for getting structured outputs out of models more generally.\n",
+    "\n",
+    "With `ChatVertexAI.bind_tools()`, we can easily pass in Pydantic classes, dict schemas, LangChain tools, or even functions as tools to the model. Under the hood these are converted to a Gemini tool schema, which looks like:\n",
+    "```python\n",
+    "{\n",
+    "    \"name\": \"...\",  # tool name\n",
+    "    \"description\": \"...\",  # tool description\n",
+    "    \"parameters\": {...}  # tool input schema as JSONSchema\n",
+    "}\n",
+    "```"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 2,
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
-       "AIMessage(content='Ich liebe Programmieren. \\n', response_metadata={'is_blocked': False, 'safety_ratings': [{'category': 'HARM_CATEGORY_HATE_SPEECH', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_DANGEROUS_CONTENT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_HARASSMENT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_SEXUALLY_EXPLICIT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}], 'usage_metadata': {'prompt_token_count': 15, 'candidates_token_count': 8, 'total_token_count': 23}}, id='run-c71955fd-8dc1-422b-88a7-853accf4811b-0', usage_metadata={'input_tokens': 15, 'output_tokens': 8, 'total_tokens': 23})"
+       "AIMessage(content='', additional_kwargs={'function_call': {'name': 'GetWeather', 'arguments': '{\"location\": \"San Francisco, CA\"}'}}, response_metadata={'is_blocked': False, 'safety_ratings': [{'category': 'HARM_CATEGORY_HATE_SPEECH', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_DANGEROUS_CONTENT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_HARASSMENT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}, {'category': 'HARM_CATEGORY_SEXUALLY_EXPLICIT', 'probability_label': 'NEGLIGIBLE', 'blocked': False}], 'citation_metadata': None, 'usage_metadata': {'prompt_token_count': 41, 'candidates_token_count': 7, 'total_token_count': 48}}, id='run-05e760dc-0682-4286-88e1-5b23df69b083-0', tool_calls=[{'name': 'GetWeather', 'args': {'location': 'San Francisco, CA'}, 'id': 'cd2499c4-4513-4059-bfff-5321b6e922d0'}])"
+      ]
+     },
+     "execution_count": 2,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "from langchain.pydantic_v1 import BaseModel, Field\n",
+    "\n",
+    "\n",
+    "class GetWeather(BaseModel):\n",
+    "    \"\"\"Get the current weather in a given location\"\"\"\n",
+    "\n",
+    "    location: str = Field(..., description=\"The city and state, e.g. San Francisco, CA\")\n",
+    "\n",
+    "\n",
+    "llm = ChatVertexAI(model=\"gemini-pro\", temperature=0)\n",
+    "llm_with_tools = llm.bind_tools([GetWeather])\n",
+    "ai_msg = llm_with_tools.invoke(\n",
+    "    \"what is the weather like in San Francisco\",\n",
+    ")\n",
+    "ai_msg"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "The tool calls can be access via the `AIMessage.tool_calls` attribute, where they are extracted in a model-agnostic format:"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 3,
+   "metadata": {},
+   "outputs": [
+    {
+     "data": {
+      "text/plain": [
+       "[{'name': 'GetWeather',\n",
+       "  'args': {'location': 'San Francisco, CA'},\n",
+       "  'id': 'cd2499c4-4513-4059-bfff-5321b6e922d0'}]"
+      ]
+     },
+     "execution_count": 3,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "ai_msg.tool_calls"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "For a complete guide on tool calling [head here](/docs/how_to/function_calling)."
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "## Structured outputs\n",
+    "\n",
+    "Many applications require structured model outputs. Tool calling makes it much easier to do this reliably. The [with_structured_outputs](https://api.python.langchain.com/en/latest/chat_models/langchain_google_vertexai.chat_models.ChatVertexAI.html) constructor provides a simple interface built on top of tool calling for getting structured outputs out of a model. For a complete guide on structured outputs [head here](/docs/how_to/structured_output).\n",
+    "\n",
+    "###  ChatVertexAI.with_structured_outputs()\n",
+    "\n",
+    "To get structured outputs from our Gemini model all we need to do is to specify a desired schema, either as a Pydantic class or as a JSON schema, "
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 6,
+   "metadata": {},
+   "outputs": [
+    {
+     "data": {
+      "text/plain": [
+       "Person(name='Stefan', age=13)"
      ]
     },
     "execution_count": 6,
@@ -212,36 +471,139 @@
    }
   ],
   "source": [
-    "from langchain_core.prompts import ChatPromptTemplate\n",
+    "class Person(BaseModel):\n",
+    "    \"\"\"Save information about a person.\"\"\"\n",
    "\n",
-    "prompt = ChatPromptTemplate.from_messages(\n",
-    "    [\n",
-    "        (\n",
-    "            \"system\",\n",
-    "            \"You are a helpful assistant that translates {input_language} to {output_language}.\",\n",
-    "        ),\n",
-    "        (\"human\", \"{input}\"),\n",
-    "    ]\n",
+    "    name: str = Field(..., description=\"The person's name.\")\n",
+    "    age: int = Field(..., description=\"The person's age.\")\n",
+    "\n",
+    "\n",
+    "structured_llm = llm.with_structured_output(Person)\n",
+    "structured_llm.invoke(\"Stefan is already 13 years old\")"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "### [Legacy] Using `create_structured_runnable()`\n",
+    "\n",
+    "The legacy wasy to get structured outputs is using the `create_structured_runnable` constructor:"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "from langchain_google_vertexai import create_structured_runnable\n",
+    "\n",
+    "chain = create_structured_runnable(Person, llm)\n",
+    "chain.invoke(\"My name is Erick and I'm 27 years old\")"
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "## Asynchronous calls\n",
+    "\n",
+    "We can make asynchronous calls via the Runnables [Async Interface](/docs/concepts#interface)."
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "# for running these examples in the notebook:\n",
+    "import asyncio\n",
+    "\n",
+    "import nest_asyncio\n",
+    "\n",
+    "nest_asyncio.apply()"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [
+    {
+     "data": {
+      "text/plain": [
+       "AIMessage(content=' अहं प्रोग्रामनं प्रेमामि')"
+      ]
+     },
+     "execution_count": null,
+     "metadata": {},
+     "output_type": "execute_result"
+    }
+   ],
+   "source": [
+    "system = (\n",
+    "    \"You are a helpful assistant that translates {input_language} to {output_language}.\"\n",
    ")\n",
+    "human = \"{text}\"\n",
+    "prompt = ChatPromptTemplate.from_messages([(\"system\", system), (\"human\", human)])\n",
    "\n",
-    "chain = prompt | llm\n",
-    "chain.invoke(\n",
-    "    {\n",
-    "        \"input_language\": \"English\",\n",
-    "        \"output_language\": \"German\",\n",
-    "        \"input\": \"I love programming.\",\n",
-    "    }\n",
+    "chat = ChatVertexAI(model=\"chat-bison\", max_tokens=1000, temperature=0.5)\n",
+    "chain = prompt | chat\n",
+    "\n",
+    "asyncio.run(\n",
+    "    chain.ainvoke(\n",
+    "        {\n",
+    "            \"input_language\": \"English\",\n",
+    "            \"output_language\": \"Sanskrit\",\n",
+    "            \"text\": \"I love programming\",\n",
+    "        }\n",
+    "    )\n",
    ")"
   ]
  },
  {
   "cell_type": "markdown",
-   "id": "3a5bb5ca-c3ae-4a58-be67-2cd18574b9a3",
   "metadata": {},
   "source": [
-    "## API reference\n",
+    "## Streaming calls\n",
    "\n",
-    "For detailed documentation of all ChatVertexAI features and configurations, like how to send multimodal inputs and configure safety settings, head to the API reference: https://api.python.langchain.com/en/latest/chat_models/langchain_google_vertexai.chat_models.ChatVertexAI.html"
+    "We can also stream outputs via the `stream` method:"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": null,
+   "metadata": {},
+   "outputs": [
+    {
+     "name": "stdout",
+     "output_type": "stream",
+     "text": [
+      " The five most populous countries in the world are:\n",
+      "1. China (1.4 billion)\n",
+      "2. India (1.3 billion)\n",
+      "3. United States (331 million)\n",
+      "4. Indonesia (273 million)\n",
+      "5. Pakistan (220 million)"
+     ]
+    }
+   ],
+   "source": [
+    "import sys\n",
+    "\n",
+    "prompt = ChatPromptTemplate.from_messages(\n",
+    "    [(\"human\", \"List out the 5 most populous countries in the world\")]\n",
+    ")\n",
+    "\n",
+    "chat = ChatVertexAI()\n",
+    "\n",
+    "chain = prompt | chat\n",
+    "\n",
+    "for chunk in chain.stream({}):\n",
+    "    sys.stdout.write(chunk.content)\n",
+    "    sys.stdout.flush()"
   ]
  }
 ],
@@ -265,5 +627,5 @@
  }
 },
 "nbformat": 4,
- "nbformat_minor": 5
+ "nbformat_minor": 4
 }
--- a/docs/docs/integrations/chat/groq.ipynb
+++ b/docs/docs/integrations/chat/groq.ipynb
@@ -2,15 +2,10 @@
 "cells": [
  {
   "cell_type": "raw",
-   "metadata": {
-    "vscode": {
-     "languageId": "raw"
-    }
-   },
+   "metadata": {},
   "source": [
    "---\n",
    "sidebar_label: Groq\n",
-    "keywords: [chatgroq]\n",
    "---"
   ]
  },
@@ -20,67 +15,45 @@
   "source": [
    "# Groq\n",
    "\n",
-    "LangChain supports integration with [Groq](https://groq.com/) chat models. Groq specializes in fast AI inference.\n",
+    "Install the langchain-groq package if not already installed:\n",
+    "\n",
+    "```bash\n",
+    "pip install langchain-groq\n",
+    "```\n",
    "\n",
-    "To get started, you'll first need to install the langchain-groq package:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "%pip install -qU langchain-groq"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
    "Request an [API key](https://wow.groq.com) and set it as an environment variable:\n",
    "\n",
    "```bash\n",
    "export GROQ_API_KEY=<YOUR API KEY>\n",
    "```\n",
    "\n",
-    "Alternatively, you may configure the API key when you initialize ChatGroq.\n",
-    "\n",
-    "Here's an example of it in action:"
+    "Alternatively, you may configure the API key when you initialize ChatGroq."
+   ]
+  },
+  {
+   "cell_type": "markdown",
+   "metadata": {},
+   "source": [
+    "Import the ChatGroq class and initialize it with a model:"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 8,
+   "execution_count": 27,
   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "AIMessage(content=\"Low latency is crucial for Large Language Models (LLMs) because it directly impacts the user experience, model performance, and overall efficiency. Here are some reasons why low latency is essential for LLMs:\\n\\n1. **Real-time Interaction**: LLMs are often used in applications that require real-time interaction, such as chatbots, virtual assistants, and language translation. Low latency ensures that the model responds quickly to user input, providing a seamless and engaging experience.\\n2. **Conversational Flow**: In conversational AI, latency can disrupt the natural flow of conversation. Low latency helps maintain a smooth conversation, allowing users to respond quickly and naturally, without feeling like they're waiting for the model to catch up.\\n3. **Model Performance**: High latency can lead to increased error rates, as the model may struggle to keep up with the input pace. Low latency enables the model to process information more efficiently, resulting in better accuracy and performance.\\n4. **Scalability**: As the number of users and requests increases, low latency becomes even more critical. It allows the model to handle a higher volume of requests without sacrificing performance, making it more scalable and efficient.\\n5. **Resource Utilization**: Low latency can reduce the computational resources required to process requests. By minimizing latency, you can optimize resource allocation, reduce costs, and improve overall system efficiency.\\n6. **User Experience**: High latency can lead to frustration, abandonment, and a poor user experience. Low latency ensures that users receive timely responses, which is essential for building trust and satisfaction.\\n7. **Competitive Advantage**: In applications like customer service or language translation, low latency can be a key differentiator. It can provide a competitive advantage by offering a faster and more responsive experience, setting your application apart from others.\\n8. **Edge Computing**: With the increasing adoption of edge computing, low latency is critical for processing data closer to the user. This reduces latency even further, enabling real-time processing and analysis of data.\\n9. **Real-time Analytics**: Low latency enables real-time analytics and insights, which are essential for applications like sentiment analysis, trend detection, and anomaly detection.\\n10. **Future-Proofing**: As LLMs continue to evolve and become more complex, low latency will become even more critical. By prioritizing low latency now, you'll be better prepared to handle the demands of future LLM applications.\\n\\nIn summary, low latency is vital for LLMs because it ensures a seamless user experience, improves model performance, and enables efficient resource utilization. By prioritizing low latency, you can build more effective, scalable, and efficient LLM applications that meet the demands of real-time interaction and processing.\", response_metadata={'token_usage': {'completion_tokens': 541, 'prompt_tokens': 33, 'total_tokens': 574, 'completion_time': 1.499777658, 'prompt_time': 0.008344704, 'queue_time': None, 'total_time': 1.508122362}, 'model_name': 'llama3-70b-8192', 'system_fingerprint': 'fp_87cbfbbc4d', 'finish_reason': 'stop', 'logprobs': None}, id='run-49dad960-ace8-4cd7-90b3-2db99ecbfa44-0')"
-      ]
-     },
-     "execution_count": 8,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
+   "outputs": [],
   "source": [
    "from langchain_core.prompts import ChatPromptTemplate\n",
-    "from langchain_groq import ChatGroq\n",
-    "\n",
-    "chat = ChatGroq(\n",
-    "    temperature=0,\n",
-    "    model=\"llama3-70b-8192\",\n",
-    "    # api_key=\"\" # Optional if not set as an environment variable\n",
-    ")\n",
-    "\n",
-    "system = \"You are a helpful assistant.\"\n",
-    "human = \"{text}\"\n",
-    "prompt = ChatPromptTemplate.from_messages([(\"system\", system), (\"human\", human)])\n",
-    "\n",
-    "chain = prompt | chat\n",
-    "chain.invoke({\"text\": \"Explain the importance of low latency for LLMs.\"})"
+    "from langchain_groq import ChatGroq"
+   ]
+  },
+  {
+   "cell_type": "code",
+   "execution_count": 28,
+   "metadata": {},
+   "outputs": [],
+   "source": [
+    "chat = ChatGroq(temperature=0, model_name=\"mixtral-8x7b-32768\")"
   ]
  },
  {
@@ -89,206 +62,97 @@
   "source": [
    "You can view the available models [here](https://console.groq.com/docs/models).\n",
    "\n",
-    "## Tool calling\n",
+    "If you do not want to set your API key in the environment, you can pass it directly to the client:\n",
+    "```python\n",
+    "chat = ChatGroq(temperature=0, groq_api_key=\"YOUR_API_KEY\", model_name=\"mixtral-8x7b-32768\")\n",
    "\n",
-    "Groq chat models support [tool calling](/docs/how_to/tool_calling/) to generate output matching a specific schema. The model may choose to call multiple tools or the same tool multiple times if appropriate.\n",
-    "\n",
-    "Here's an example:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 10,
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[{'name': 'get_current_weather',\n",
-       "  'args': {'location': 'San Francisco', 'unit': 'Celsius'},\n",
-       "  'id': 'call_pydj'},\n",
-       " {'name': 'get_current_weather',\n",
-       "  'args': {'location': 'Tokyo', 'unit': 'Celsius'},\n",
-       "  'id': 'call_jgq3'}]"
-      ]
-     },
-     "execution_count": 10,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "from typing import Optional\n",
-    "\n",
-    "from langchain_core.tools import tool\n",
-    "\n",
-    "\n",
-    "@tool\n",
-    "def get_current_weather(location: str, unit: Optional[str]):\n",
-    "    \"\"\"Get the current weather in a given location\"\"\"\n",
-    "    return \"Cloudy with a chance of rain.\"\n",
-    "\n",
-    "\n",
-    "tool_model = chat.bind_tools([get_current_weather], tool_choice=\"auto\")\n",
-    "\n",
-    "res = tool_model.invoke(\"What is the weather like in San Francisco and Tokyo?\")\n",
-    "\n",
-    "res.tool_calls"
+    "```"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
-    "### `.with_structured_output()`\n",
-    "\n",
-    "You can also use the convenience [`.with_structured_output()`](/docs/how_to/structured_output/#the-with_structured_output-method) method to coerce `ChatGroq` into returning a structured output.\n",
-    "Here is an example:"
+    "Write a prompt and invoke ChatGroq to create completions:"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 11,
+   "execution_count": 29,
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
-       "Joke(setup='Why did the cat join a band?', punchline='Because it wanted to be the purr-cussionist!', rating=None)"
+       "AIMessage(content='Low Latency Large Language Models (LLMs) are a type of artificial intelligence model that can understand and generate human-like text. The term \"low latency\" refers to the model\\'s ability to process and respond to inputs quickly, with minimal delay.\\n\\nThe importance of low latency in LLMs can be explained through the following points:\\n\\n1. Improved user experience: In real-time applications such as chatbots, virtual assistants, and interactive games, users expect quick and responsive interactions. Low latency LLMs can provide instant feedback and responses, creating a more seamless and engaging user experience.\\n\\n2. Better decision-making: In time-sensitive scenarios, such as financial trading or autonomous vehicles, low latency LLMs can quickly process and analyze vast amounts of data, enabling faster and more informed decision-making.\\n\\n3. Enhanced accessibility: For individuals with disabilities, low latency LLMs can help create more responsive and inclusive interfaces, such as voice-controlled assistants or real-time captioning systems.\\n\\n4. Competitive advantage: In industries where real-time data analysis and decision-making are crucial, low latency LLMs can provide a competitive edge by enabling businesses to react more quickly to market changes, customer needs, or emerging opportunities.\\n\\n5. Scalability: Low latency LLMs can efficiently handle a higher volume of requests and interactions, making them more suitable for large-scale applications and services.\\n\\nIn summary, low latency is an essential aspect of LLMs, as it significantly impacts user experience, decision-making, accessibility, competitiveness, and scalability. By minimizing delays and response times, low latency LLMs can unlock new possibilities and applications for artificial intelligence in various industries and scenarios.')"
      ]
     },
-     "execution_count": 11,
+     "execution_count": 29,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
-    "from langchain_core.pydantic_v1 import BaseModel, Field\n",
+    "system = \"You are a helpful assistant.\"\n",
+    "human = \"{text}\"\n",
+    "prompt = ChatPromptTemplate.from_messages([(\"system\", system), (\"human\", human)])\n",
    "\n",
-    "\n",
-    "class Joke(BaseModel):\n",
-    "    \"\"\"Joke to tell user.\"\"\"\n",
-    "\n",
-    "    setup: str = Field(description=\"The setup of the joke\")\n",
-    "    punchline: str = Field(description=\"The punchline to the joke\")\n",
-    "    rating: Optional[int] = Field(description=\"How funny the joke is, from 1 to 10\")\n",
-    "\n",
-    "\n",
-    "structured_llm = chat.with_structured_output(Joke)\n",
-    "\n",
-    "structured_llm.invoke(\"Tell me a joke about cats\")"
+    "chain = prompt | chat\n",
+    "chain.invoke({\"text\": \"Explain the importance of low latency LLMs.\"})"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
-    "Behind the scenes, this takes advantage of the above tool calling functionality.\n",
-    "\n",
-    "## Async"
+    "## `ChatGroq` also supports async and streaming functionality:"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 12,
+   "execution_count": 32,
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
-       "AIMessage(content='Here is a limerick about the sun:\\n\\nThere once was a sun in the sky,\\nWhose warmth and light caught the eye,\\nIt shone bright and bold,\\nWith a fiery gold,\\nAnd brought life to all, as it flew by.', response_metadata={'token_usage': {'completion_tokens': 51, 'prompt_tokens': 18, 'total_tokens': 69, 'completion_time': 0.144614022, 'prompt_time': 0.00585394, 'queue_time': None, 'total_time': 0.150467962}, 'model_name': 'llama3-70b-8192', 'system_fingerprint': 'fp_2f30b0b571', 'finish_reason': 'stop', 'logprobs': None}, id='run-e42340ba-f0ad-4b54-af61-8308d8ec8256-0')"
+       "AIMessage(content=\"There's a star that shines up in the sky,\\nThe Sun, that makes the day bright and spry.\\nIt rises and sets,\\nIn a daily, predictable bet,\\nGiving life to the world, oh my!\")"
      ]
     },
-     "execution_count": 12,
+     "execution_count": 32,
     "metadata": {},
     "output_type": "execute_result"
    }
   ],
   "source": [
-    "chat = ChatGroq(temperature=0, model=\"llama3-70b-8192\")\n",
+    "chat = ChatGroq(temperature=0, model_name=\"mixtral-8x7b-32768\")\n",
    "prompt = ChatPromptTemplate.from_messages([(\"human\", \"Write a Limerick about {topic}\")])\n",
    "chain = prompt | chat\n",
    "await chain.ainvoke({\"topic\": \"The Sun\"})"
   ]
  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "## Streaming"
-   ]
-  },
  {
   "cell_type": "code",
-   "execution_count": 13,
+   "execution_count": 33,
   "metadata": {},
   "outputs": [
    {
     "name": "stdout",
     "output_type": "stream",
     "text": [
-      "Silvery glow bright\n",
-      "Luna's gentle light shines down\n",
-      "Midnight's gentle queen"
+      "The moon's gentle glow\n",
+      "Illuminates the night sky\n",
+      "Peaceful and serene"
     ]
    }
   ],
   "source": [
-    "chat = ChatGroq(temperature=0, model=\"llama3-70b-8192\")\n",
+    "chat = ChatGroq(temperature=0, model_name=\"llama2-70b-4096\")\n",
    "prompt = ChatPromptTemplate.from_messages([(\"human\", \"Write a haiku about {topic}\")])\n",
    "chain = prompt | chat\n",
    "for chunk in chain.stream({\"topic\": \"The Moon\"}):\n",
    "    print(chunk.content, end=\"\", flush=True)"
   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "## Passing custom parameters\n",
-    "\n",
-    "You can pass other Groq-specific parameters using the `model_kwargs` argument on initialization. Here's an example of enabling JSON mode:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 15,
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "AIMessage(content='{ \"response\": \"That\\'s a tough question! There are eight species of bears found in the world, and each one is unique and amazing in its own way. However, if I had to pick one, I\\'d say the giant panda is a popular favorite among many people. Who can resist those adorable black and white markings?\", \"followup_question\": \"Would you like to know more about the giant panda\\'s habitat and diet?\" }', response_metadata={'token_usage': {'completion_tokens': 89, 'prompt_tokens': 50, 'total_tokens': 139, 'completion_time': 0.249032839, 'prompt_time': 0.011134497, 'queue_time': None, 'total_time': 0.260167336}, 'model_name': 'llama3-70b-8192', 'system_fingerprint': 'fp_2f30b0b571', 'finish_reason': 'stop', 'logprobs': None}, id='run-558ce67e-8c63-43fe-a48f-6ecf181bc922-0')"
-      ]
-     },
-     "execution_count": 15,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "chat = ChatGroq(\n",
-    "    model=\"llama3-70b-8192\", model_kwargs={\"response_format\": {\"type\": \"json_object\"}}\n",
-    ")\n",
-    "\n",
-    "system = \"\"\"\n",
-    "You are a helpful assistant.\n",
-    "Always respond with a JSON object with two string keys: \"response\" and \"followup_question\".\n",
-    "\"\"\"\n",
-    "human = \"{question}\"\n",
-    "prompt = ChatPromptTemplate.from_messages([(\"system\", system), (\"human\", human)])\n",
-    "\n",
-    "chain = prompt | chat\n",
-    "\n",
-    "chain.invoke({\"question\": \"what bear is best?\"})"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": []
  }
 ],
 "metadata": {
@@ -307,7 +171,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
-   "version": "3.10.5"
+   "version": "3.10.13"
  }
 },
 "nbformat": 4,
--- a/docs/docs/integrations/chat/huggingface.ipynb
+++ b/docs/docs/integrations/chat/huggingface.ipynb
@@ -315,11 +315,7 @@
   "source": [
    "## 4. Take it for a spin as an agent!\n",
    "\n",
-    "Here we'll test out `Zephyr-7B-beta` as a zero-shot `ReAct` Agent. \n",
-    "\n",
-    "The agent is based on the paper [ReAct: Synergizing Reasoning and Acting in Language Models](https://arxiv.org/abs/2210.03629)\n",
-    "\n",
-    "The example below is taken from [here](https://python.langchain.com/v0.1/docs/modules/agents/agent_types/react/#using-chat-models).\n",
+    "Here we'll test out `Zephyr-7B-beta` as a zero-shot `ReAct` Agent. The example below is taken from [here](https://python.langchain.com/v0.1/docs/modules/agents/agent_types/react/#using-chat-models).\n",
    "\n",
    "> Note: To run this section, you'll need to have a [SerpAPI Token](https://serpapi.com/) saved as an environment variable: `SERPAPI_API_KEY`"
   ]
--- a/docs/docs/integrations/chat/llamacpp.ipynb
+++ b/docs/docs/integrations/chat/llamacpp.ipynb
--- a/docs/docs/integrations/chat/mistralai.ipynb
+++ b/docs/docs/integrations/chat/mistralai.ipynb
@@ -225,7 +225,7 @@
   "source": [
    "## Chaining\n",
    "\n",
-    "You can also easily combine with a prompt template for easy structuring of user input. We can do this using [LCEL](/docs/concepts#langchain-expression-language-lcel)"
+    "You can also easily combine with a prompt template for easy structuring of user input. We can do this using [LCEL](/docs/concepts#langchain-expression-language)"
   ]
  },
  {
--- a/docs/docs/integrations/chat/nvidia_ai_endpoints.ipynb
+++ b/docs/docs/integrations/chat/nvidia_ai_endpoints.ipynb
@@ -134,7 +134,7 @@
    "from langchain_nvidia_ai_endpoints import ChatNVIDIA\n",
    "\n",
    "# connect to an embedding NIM running at localhost:8000, specifying a specific model\n",
-    "llm = ChatNVIDIA(base_url=\"http://localhost:8000/v1\", model=\"meta/llama3-8b-instruct\")"
+    "llm = ChatNVIDIA(base_url=\"http://localhost:8000/v1\", model=\"meta-llama3-8b-instruct\")"
   ]
  },
  {
@@ -658,7 +658,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
-   "version": "3.10.2"
+   "version": "3.10.13"
  }
 },
 "nbformat": 4,
--- a/docs/docs/integrations/chat/premai.ipynb
+++ b/docs/docs/integrations/chat/premai.ipynb
@@ -238,67 +238,6 @@
    "> Ideally, you do not need to connect Repository IDs here to get Retrieval Augmented Generations. You can still get the same result if you have connected the repositories in prem platform. "
   ]
  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "### Prem Templates\n",
-    "\n",
-    "Writing Prompt Templates can be super messy. Prompt templates are long, hard to manage, and must be continuously tweaked to improve and keep the same throughout the application. \n",
-    "\n",
-    "With **Prem**, writing and managing prompts can be super easy. The **_Templates_** tab inside the [launchpad](https://docs.premai.io/get-started/launchpad) helps you write as many prompts you need and use it inside the SDK to make your application running using those prompts. You can read more about Prompt Templates [here](https://docs.premai.io/get-started/prem-templates). \n",
-    "\n",
-    "To use Prem Templates natively with LangChain, you need to pass an id the `HumanMessage`. This id should be the name the variable of your prompt template. the `content` in `HumanMessage` should be the value of that variable. \n",
-    "\n",
-    "let's say for example, if your prompt template was this:\n",
-    "\n",
-    "```text\n",
-    "Say hello to my name and say a feel-good quote\n",
-    "from my age. My name is: {name} and age is {age}\n",
-    "```\n",
-    "\n",
-    "So now your human_messages should look like:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "human_messages = [\n",
-    "    HumanMessage(content=\"Shawn\", id=\"name\"),\n",
-    "    HumanMessage(content=\"22\", id=\"age\"),\n",
-    "]"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "\n",
-    "Pass this `human_messages` to ChatPremAI Client. Please note: Do not forget to\n",
-    "pass the additional `template_id` to invoke generation with Prem Templates. If you are not aware of `template_id` you can learn more about that [in our docs](https://docs.premai.io/get-started/prem-templates). Here is an example:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "template_id = \"78069ce8-xxxxx-xxxxx-xxxx-xxx\"\n",
-    "response = chat.invoke([human_message], template_id=template_id)\n",
-    "print(response.content)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "Prem Template feature is available in streaming too. "
-   ]
-  },
  {
   "cell_type": "markdown",
   "metadata": {},
--- a/docs/docs/integrations/chat/snowflake.ipynb
+++ b/docs/docs/integrations/chat/snowflake.ipynb
@@ -1,180 +0,0 @@
-{
- "cells": [
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "# Snowflake Cortex\n",
-    "\n",
-    "[Snowflake Cortex](https://docs.snowflake.com/en/user-guide/snowflake-cortex/llm-functions) gives you instant access to industry-leading large language models (LLMs) trained by researchers at companies like Mistral, Reka, Meta, and Google, including [Snowflake Arctic](https://www.snowflake.com/en/data-cloud/arctic/), an open enterprise-grade model developed by Snowflake.\n",
-    "\n",
-    "This example goes over how to use LangChain to interact with Snowflake Cortex."
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "### Installation and setup\n",
-    "\n",
-    "We start by installing the `snowflake-snowpark-python` library, using the command below. Then we configure the credentials for connecting to Snowflake, as environment variables or pass them directly."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 1,
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "Note: you may need to restart the kernel to use updated packages.\n"
-     ]
-    }
-   ],
-   "source": [
-    "%pip install --upgrade --quiet snowflake-snowpark-python"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 2,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "import getpass\n",
-    "import os\n",
-    "\n",
-    "# First step is to set up the environment variables, to connect to Snowflake,\n",
-    "# you can also pass these snowflake credentials while instantiating the model\n",
-    "\n",
-    "if os.environ.get(\"SNOWFLAKE_ACCOUNT\") is None:\n",
-    "    os.environ[\"SNOWFLAKE_ACCOUNT\"] = getpass.getpass(\"Account: \")\n",
-    "\n",
-    "if os.environ.get(\"SNOWFLAKE_USERNAME\") is None:\n",
-    "    os.environ[\"SNOWFLAKE_USERNAME\"] = getpass.getpass(\"Username: \")\n",
-    "\n",
-    "if os.environ.get(\"SNOWFLAKE_PASSWORD\") is None:\n",
-    "    os.environ[\"SNOWFLAKE_PASSWORD\"] = getpass.getpass(\"Password: \")\n",
-    "\n",
-    "if os.environ.get(\"SNOWFLAKE_DATABASE\") is None:\n",
-    "    os.environ[\"SNOWFLAKE_DATABASE\"] = getpass.getpass(\"Database: \")\n",
-    "\n",
-    "if os.environ.get(\"SNOWFLAKE_SCHEMA\") is None:\n",
-    "    os.environ[\"SNOWFLAKE_SCHEMA\"] = getpass.getpass(\"Schema: \")\n",
-    "\n",
-    "if os.environ.get(\"SNOWFLAKE_WAREHOUSE\") is None:\n",
-    "    os.environ[\"SNOWFLAKE_WAREHOUSE\"] = getpass.getpass(\"Warehouse: \")\n",
-    "\n",
-    "if os.environ.get(\"SNOWFLAKE_ROLE\") is None:\n",
-    "    os.environ[\"SNOWFLAKE_ROLE\"] = getpass.getpass(\"Role: \")"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 8,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "from langchain_community.chat_models import ChatSnowflakeCortex\n",
-    "from langchain_core.messages import HumanMessage, SystemMessage\n",
-    "\n",
-    "# By default, we'll be using the cortex provided model: `snowflake-arctic`, with function: `complete`\n",
-    "chat = ChatSnowflakeCortex()"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "The above cell assumes that your Snowflake credentials are set in your environment variables. If you would rather manually specify them, use the following code:\n",
-    "\n",
-    "```python\n",
-    "chat = ChatSnowflakeCortex(\n",
-    "    # change default cortex model and function\n",
-    "    model=\"snowflake-arctic\",\n",
-    "    cortex_function=\"complete\",\n",
-    "\n",
-    "    # change default generation parameters\n",
-    "    temperature=0,\n",
-    "    max_tokens=10,\n",
-    "    top_p=0.95,\n",
-    "\n",
-    "    # specify snowflake credentials\n",
-    "    account=\"YOUR_SNOWFLAKE_ACCOUNT\",\n",
-    "    username=\"YOUR_SNOWFLAKE_USERNAME\",\n",
-    "    password=\"YOUR_SNOWFLAKE_PASSWORD\",\n",
-    "    database=\"YOUR_SNOWFLAKE_DATABASE\",\n",
-    "    schema=\"YOUR_SNOWFLAKE_SCHEMA\",\n",
-    "    role=\"YOUR_SNOWFLAKE_ROLE\",\n",
-    "    warehouse=\"YOUR_SNOWFLAKE_WAREHOUSE\"\n",
-    ")\n",
-    "```"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "### Calling the model\n",
-    "We can now call the model using the `invoke` or `generate` method.\n",
-    "\n",
-    "#### Generation"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 9,
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "AIMessage(content=\" Large language models are artificial intelligence systems designed to understand, generate, and manipulate human language. These models are typically based on deep learning techniques and are trained on vast amounts of text data to learn patterns and structures in language. They can perform a wide range of language-related tasks, such as language translation, text generation, sentiment analysis, and answering questions. Some well-known large language models include Google's BERT, OpenAI's GPT series, and Facebook's RoBERTa. These models have shown remarkable performance in various natural language processing tasks, and their applications continue to expand as research in AI progresses.\", response_metadata={'completion_tokens': 131, 'prompt_tokens': 29, 'total_tokens': 160}, id='run-5435bd0a-83fd-4295-b237-66cbd1b5c0f3-0')"
-      ]
-     },
-     "execution_count": 9,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "messages = [\n",
-    "    SystemMessage(content=\"You are a friendly assistant.\"),\n",
-    "    HumanMessage(content=\"What are large language models?\"),\n",
-    "]\n",
-    "chat.invoke(messages)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "### Streaming\n",
-    "`ChatSnowflakeCortex` doesn't support streaming as of now. Support for streaming will be coming in the later versions!"
-   ]
-  }
- ],
- "metadata": {
-  "kernelspec": {
-   "display_name": ".venv",
-   "language": "python",
-   "name": "python3"
-  },
-  "language_info": {
-   "codemirror_mode": {
-    "name": "ipython",
-    "version": 3
-   },
-   "file_extension": ".py",
-   "mimetype": "text/x-python",
-   "name": "python",
-   "nbconvert_exporter": "python",
-   "pygments_lexer": "ipython3",
-   "version": "3.11.9"
-  }
- },
- "nbformat": 4,
- "nbformat_minor": 2
-}
--- a/docs/docs/integrations/document_loaders/apify_dataset.ipynb
+++ b/docs/docs/integrations/document_loaders/apify_dataset.ipynb
@@ -13,7 +13,7 @@
    "\n",
    "## Prerequisites\n",
    "\n",
-    "You need to have an existing dataset on the Apify platform. If you don't have one, please first check out [this notebook](/docs/integrations/tools/apify) on how to use Apify to extract content from documentation, knowledge bases, help centers, or blogs. This example shows how to load a dataset produced by the [Website Content Crawler](https://apify.com/apify/website-content-crawler)."
+    "You need to have an existing dataset on the Apify platform. If you don't have one, please first check out [this notebook](/docs/integrations/tools/apify) on how to use Apify to extract content from documentation, knowledge bases, help centers, or blogs."
   ]
  },
  {
@@ -101,10 +101,8 @@
   "outputs": [],
   "source": [
    "from langchain.indexes import VectorstoreIndexCreator\n",
-    "from langchain_community.utilities import ApifyWrapper\n",
-    "from langchain_core.documents import Document\n",
-    "from langchain_openai import OpenAI\n",
-    "from langchain_openai.embeddings import OpenAIEmbeddings"
+    "from langchain_community.document_loaders import ApifyDatasetLoader\n",
+    "from langchain_core.documents import Document"
   ]
  },
  {
@@ -127,7 +125,7 @@
   "metadata": {},
   "outputs": [],
   "source": [
-    "index = VectorstoreIndexCreator(embedding=OpenAIEmbeddings()).from_loaders([loader])"
+    "index = VectorstoreIndexCreator().from_loaders([loader])"
   ]
  },
  {
@@ -137,7 +135,7 @@
   "outputs": [],
   "source": [
    "query = \"What is Apify?\"\n",
-    "result = index.query_with_sources(query, llm=OpenAI())"
+    "result = index.query_with_sources(query)"
   ]
  },
  {
--- a/docs/docs/integrations/document_loaders/image_captions.ipynb
+++ b/docs/docs/integrations/document_loaders/image_captions.ipynb
@@ -83,7 +83,7 @@
   },
   "outputs": [],
   "source": [
-    "loader = ImageCaptionLoader(images=list_image_urls)\n",
+    "loader = ImageCaptionLoader(path_images=list_image_urls)\n",
    "list_docs = loader.load()\n",
    "list_docs"
   ]
--- a/docs/docs/integrations/document_loaders/kinetica.ipynb
+++ b/docs/docs/integrations/document_loaders/kinetica.ipynb
@@ -15,7 +15,7 @@
   "metadata": {},
   "outputs": [],
   "source": [
-    "%pip install gpudb==7.2.0.9"
+    "%pip install gpudb==7.2.0.1"
   ]
  },
  {
@@ -97,14 +97,14 @@
    "# data and the `SCHEMA.TABLE` combination must exist in Kinetica.\n",
    "\n",
    "QUERY = \"select text, survey_id as source from SCHEMA.TABLE limit 10\"\n",
-    "kl = KineticaLoader(\n",
+    "snowflake_loader = KineticaLoader(\n",
    "    query=QUERY,\n",
    "    host=HOST,\n",
    "    username=USERNAME,\n",
    "    password=PASSWORD,\n",
    "    metadata_columns=[\"source\"],\n",
    ")\n",
-    "kinetica_documents = kl.load()\n",
+    "kinetica_documents = snowflake_loader.load()\n",
    "print(kinetica_documents)"
   ]
  }
--- a/docs/docs/integrations/document_loaders/recursive_url.ipynb
+++ b/docs/docs/integrations/document_loaders/recursive_url.ipynb
@@ -7,132 +7,80 @@
   "source": [
    "# Recursive URL\n",
    "\n",
-    "The `RecursiveUrlLoader` lets you recursively scrape all child links from a root URL and parse them into Documents."
+    "We may want to process load all URLs under a root directory.\n",
+    "\n",
+    "For example, let's look at the [Python 3.9 Document](https://docs.python.org/3.9/).\n",
+    "\n",
+    "This has many interesting child pages that we may want to read in bulk.\n",
+    "\n",
+    "Of course, the `WebBaseLoader` can load a list of pages. \n",
+    "\n",
+    "But, the challenge is traversing the tree of child pages and actually assembling that list!\n",
+    " \n",
+    "We do this using the `RecursiveUrlLoader`.\n",
+    "\n",
+    "This also gives us the flexibility to exclude some children, customize the extractor, and more."
   ]
  },
  {
   "cell_type": "markdown",
-   "id": "947d29e7-3679-483d-973f-79ea3403a370",
+   "id": "1be8094f",
   "metadata": {},
   "source": [
-    "## Setup\n",
-    "\n",
-    "The `RecursiveUrlLoader` lives in the `langchain-community` package. There's no other required packages, though you will get richer default Document metadata if you have ``beautifulsoup4` installed as well."
+    "# Parameters\n",
+    "- url: str, the target url to crawl.\n",
+    "- exclude_dirs: Optional[str], webpage directories to exclude.\n",
+    "- use_async: Optional[bool], wether to use async requests, using async requests is usually faster in large tasks. However, async will disable the lazy loading feature(the function still works, but it is not lazy). By default, it is set to False.\n",
+    "- extractor: Optional[Callable[[str], str]], a function to extract the text of the document from the webpage, by default it returns the page as it is. It is recommended to use tools like goose3 and beautifulsoup to extract the text. By default, it just returns the page as it is.\n",
+    "- max_depth: Optional[int] = None, the maximum depth to crawl. By default, it is set to 2. If you need to crawl the whole website, set it to a number that is large enough would simply do the job.\n",
+    "- timeout: Optional[int] = None, the timeout for each request, in the unit of seconds. By default, it is set to 10.\n",
+    "- prevent_outside: Optional[bool] = None, whether to prevent crawling outside the root url. By default, it is set to True."
   ]
  },
  {
   "cell_type": "code",
   "execution_count": null,
-   "id": "23359ab0-8056-4dee-8bff-c38dc079f17f",
+   "id": "23c18539",
   "metadata": {},
   "outputs": [],
   "source": [
-    "%pip install -qU langchain-community beautifulsoup4"
+    "from langchain_community.document_loaders.recursive_url_loader import RecursiveUrlLoader"
   ]
  },
  {
   "cell_type": "markdown",
-   "id": "07985766-e4e9-4ea1-8a18-924fa4f294e5",
+   "id": "6384c057",
   "metadata": {},
   "source": [
-    "## Instantiation\n",
-    "\n",
-    "Now we can instantiate our document loader object and load Documents:"
+    "Let's try a simple example."
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 1,
-   "id": "cb208dcf-9ce9-4197-bc44-b80d20aa4e50",
+   "execution_count": null,
+   "id": "55394afe",
   "metadata": {},
   "outputs": [],
   "source": [
-    "from langchain_community.document_loaders import RecursiveUrlLoader\n",
+    "from bs4 import BeautifulSoup as Soup\n",
    "\n",
+    "url = \"https://docs.python.org/3.9/\"\n",
    "loader = RecursiveUrlLoader(\n",
-    "    \"https://docs.python.org/3.9/\",\n",
-    "    # max_depth=2,\n",
-    "    # use_async=False,\n",
-    "    # extractor=None,\n",
-    "    # metadata_extractor=None,\n",
-    "    # exclude_dirs=(),\n",
-    "    # timeout=10,\n",
-    "    # check_response_status=True,\n",
-    "    # continue_on_failure=True,\n",
-    "    # prevent_outside=True,\n",
-    "    # base_url=None,\n",
-    "    # ...\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "0fac4425-735f-487d-a12b-c8ed2a209039",
-   "metadata": {},
-   "source": [
-    "## Load\n",
-    "\n",
-    "Use ``.load()`` to synchronously load into memory all Documents, with one\n",
-    "Document per visited URL. Starting from the initial URL, we recurse through\n",
-    "all linked URLs up to the specified max_depth.\n",
-    "\n",
-    "Let's run through a basic example of how to use the `RecursiveUrlLoader` on the [Python 3.9 Documentation](https://docs.python.org/3.9/)."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 2,
-   "id": "a30843c8-4a59-43dc-bf60-f26532f0f8e1",
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stderr",
-     "output_type": "stream",
-     "text": [
-      "/Users/bagatur/.pyenv/versions/3.9.1/lib/python3.9/html/parser.py:170: XMLParsedAsHTMLWarning: It looks like you're parsing an XML document using an HTML parser. If this really is an HTML document (maybe it's XHTML?), you can ignore or filter this warning. If it's XML, you should know that using an XML parser will be more reliable. To parse this document as XML, make sure you have the lxml package installed, and pass the keyword argument `features=\"xml\"` into the BeautifulSoup constructor.\n",
-      "  k = self.parse_starttag(i)\n"
-     ]
-    },
-    {
-     "data": {
-      "text/plain": [
-       "{'source': 'https://docs.python.org/3.9/',\n",
-       " 'content_type': 'text/html',\n",
-       " 'title': '3.9.19 Documentation',\n",
-       " 'language': None}"
-      ]
-     },
-     "execution_count": 2,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "docs = loader.load()\n",
-    "docs[0].metadata"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "211856ed-6dd7-46c6-859e-11aaea9093db",
-   "metadata": {},
-   "source": [
-    "Great! The first document looks like the root page we started from. Let's look at the metadata of the next document"
+    "    url=url, max_depth=2, extractor=lambda x: Soup(x, \"html.parser\").text\n",
+    ")\n",
+    "docs = loader.load()"
   ]
  },
  {
   "cell_type": "code",
   "execution_count": 3,
-   "id": "2d842c03-fab8-4097-9f4f-809b2e71c0ba",
+   "id": "084fb2ce",
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
-       "{'source': 'https://docs.python.org/3.9/using/index.html',\n",
-       " 'content_type': 'text/html',\n",
-       " 'title': 'Python Setup and Usage — Python 3.9.19 documentation',\n",
-       " 'language': None}"
+       "'\\n\\n\\n\\n\\nPython Frequently Asked Questions — Python 3.'"
      ]
     },
     "execution_count": 3,
@@ -141,175 +89,70 @@
    }
   ],
   "source": [
-    "docs[1].metadata"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "f5714ace-7cc5-4c5c-9426-f68342880da0",
-   "metadata": {},
-   "source": [
-    "That url looks like a child of our root page, which is great! Let's move on from metadata to examine the content of one of our documents"
+    "docs[0].page_content[:50]"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 6,
-   "id": "51dc6c67-6857-4298-9472-08b147f3a631",
+   "execution_count": 4,
+   "id": "13bd7e16",
   "metadata": {},
   "outputs": [
    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "\n",
-      "<!DOCTYPE html>\n",
-      "\n",
-      "<html xmlns=\"http://www.w3.org/1999/xhtml\">\n",
-      "  <head>\n",
-      "    <meta charset=\"utf-8\" /><title>3.9.19 Documentation</title><meta name=\"viewport\" content=\"width=device-width, initial-scale=1.0\">\n",
-      "    \n",
-      "    <link rel=\"stylesheet\" href=\"_static/pydoctheme.css\" type=\"text/css\" />\n",
-      "    <link rel=\n"
-     ]
+     "data": {
+      "text/plain": [
+       "{'source': 'https://docs.python.org/3.9/library/index.html',\n",
+       " 'title': 'The Python Standard Library — Python 3.9.17 documentation',\n",
+       " 'language': None}"
+      ]
+     },
+     "execution_count": 4,
+     "metadata": {},
+     "output_type": "execute_result"
    }
   ],
   "source": [
-    "print(docs[0].page_content[:300])"
+    "docs[-1].metadata"
   ]
  },
  {
   "cell_type": "markdown",
-   "id": "d87cc239",
+   "id": "5866e5a6",
   "metadata": {},
   "source": [
-    "That certainly looks like HTML that comes from the url https://docs.python.org/3.9/, which is what we expected. Let's now look at some variations we can make to our basic example that can be helpful in different situations. "
+    "However, since it's hard to perform a perfect filter, you may still see some irrelevant results in the results. You can perform a filter on the returned documents by yourself, if it's needed. Most of the time, the returned results are good enough."
   ]
  },
  {
   "cell_type": "markdown",
-   "id": "8f41cc89",
+   "id": "4ec8ecef",
   "metadata": {},
   "source": [
-    "## Adding an Extractor\n",
-    "\n",
-    "By default the loader sets the raw HTML from each link as the Document page content. To parse this HTML into a more human/LLM-friendly format you can pass in a custom ``extractor`` method:"
+    "Testing on LangChain docs."
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 21,
-   "id": "33a6f5b8",
+   "execution_count": 2,
+   "id": "349b5598",
   "metadata": {},
   "outputs": [
    {
-     "name": "stderr",
-     "output_type": "stream",
-     "text": [
-      "/var/folders/td/vzm913rx77x21csd90g63_7c0000gn/T/ipykernel_10935/1083427287.py:6: XMLParsedAsHTMLWarning: It looks like you're parsing an XML document using an HTML parser. If this really is an HTML document (maybe it's XHTML?), you can ignore or filter this warning. If it's XML, you should know that using an XML parser will be more reliable. To parse this document as XML, make sure you have the lxml package installed, and pass the keyword argument `features=\"xml\"` into the BeautifulSoup constructor.\n",
-      "  soup = BeautifulSoup(html, \"lxml\")\n",
-      "/Users/isaachershenson/.pyenv/versions/3.11.9/lib/python3.11/html/parser.py:170: XMLParsedAsHTMLWarning: It looks like you're parsing an XML document using an HTML parser. If this really is an HTML document (maybe it's XHTML?), you can ignore or filter this warning. If it's XML, you should know that using an XML parser will be more reliable. To parse this document as XML, make sure you have the lxml package installed, and pass the keyword argument `features=\"xml\"` into the BeautifulSoup constructor.\n",
-      "  k = self.parse_starttag(i)\n"
-     ]
-    },
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "3.9.19 Documentation\n",
-      "\n",
-      "Download\n",
-      "Download these documents\n",
-      "Docs by version\n",
-      "\n",
-      "Python 3.13 (in development)\n",
-      "Python 3.12 (stable)\n",
-      "Python 3.11 (security-fixes)\n",
-      "Python 3.10 (security-fixes)\n",
-      "Python 3.9 (securit\n"
-     ]
+     "data": {
+      "text/plain": [
+       "8"
+      ]
+     },
+     "execution_count": 2,
+     "metadata": {},
+     "output_type": "execute_result"
    }
   ],
   "source": [
-    "import re\n",
-    "\n",
-    "from bs4 import BeautifulSoup\n",
-    "\n",
-    "\n",
-    "def bs4_extractor(html: str) -> str:\n",
-    "    soup = BeautifulSoup(html, \"lxml\")\n",
-    "    return re.sub(r\"\\n\\n+\", \"\\n\\n\", soup.text).strip()\n",
-    "\n",
-    "\n",
-    "loader = RecursiveUrlLoader(\"https://docs.python.org/3.9/\", extractor=bs4_extractor)\n",
+    "url = \"https://js.langchain.com/docs/modules/memory/integrations/\"\n",
+    "loader = RecursiveUrlLoader(url=url)\n",
    "docs = loader.load()\n",
-    "print(docs[0].page_content[:200])"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "c8e8a826",
-   "metadata": {},
-   "source": [
-    "This looks much nicer!\n",
-    "\n",
-    "You can similarly pass in a `metadata_extractor` to customize how Document metadata is extracted from the HTTP response. See the [API reference](https://api.python.langchain.com/en/latest/document_loaders/langchain_community.document_loaders.recursive_url_loader.RecursiveUrlLoader.html) for more on this."
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "1dddbc94",
-   "metadata": {},
-   "source": [
-    "## Lazy loading\n",
-    "\n",
-    "If we're loading a  large number of Documents and our downstream operations can be done over subsets of all loaded Documents, we can lazily load our Documents one at a time to minimize our memory footprint:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 15,
-   "id": "7d0114fc",
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stderr",
-     "output_type": "stream",
-     "text": [
-      "/var/folders/4j/2rz3865x6qg07tx43146py8h0000gn/T/ipykernel_73962/2110507528.py:6: XMLParsedAsHTMLWarning: It looks like you're parsing an XML document using an HTML parser. If this really is an HTML document (maybe it's XHTML?), you can ignore or filter this warning. If it's XML, you should know that using an XML parser will be more reliable. To parse this document as XML, make sure you have the lxml package installed, and pass the keyword argument `features=\"xml\"` into the BeautifulSoup constructor.\n",
-      "  soup = BeautifulSoup(html, \"lxml\")\n"
-     ]
-    }
-   ],
-   "source": [
-    "page = []\n",
-    "for doc in loader.lazy_load():\n",
-    "    page.append(doc)\n",
-    "    if len(page) >= 10:\n",
-    "        # do some paged operation, e.g.\n",
-    "        # index.upsert(page)\n",
-    "\n",
-    "        page = []"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "f88a7c2f-35df-4c3a-b238-f91be2674b96",
-   "metadata": {},
-   "source": [
-    "In this example we never have more than 10 Documents loaded into memory at a time."
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "3e4d1c8f",
-   "metadata": {},
-   "source": [
-    "## API reference\n",
-    "\n",
-    "These examples show just a few of the ways in which you can modify the default `RecursiveUrlLoader`, but there are many more modifications that can be made to best fit your use case. Using the parameters `link_regex` and `exclude_dirs` can help you filter out unwanted URLs, `aload()` and `alazy_load()` can be used for aynchronous loading, and more.\n",
-    "\n",
-    "For detailed information on configuring and calling the ``RecursiveUrlLoader``, please see the API reference: https://api.python.langchain.com/en/latest/document_loaders/langchain_community.document_loaders.recursive_url_loader.RecursiveUrlLoader.html."
+    "len(docs)"
   ]
  }
 ],
@@ -329,7 +172,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
-   "version": "3.11.9"
+   "version": "3.10.12"
  }
 },
 "nbformat": 4,
--- a/docs/docs/integrations/document_loaders/source_code.ipynb
+++ b/docs/docs/integrations/document_loaders/source_code.ipynb
@@ -17,7 +17,6 @@
    "- C++ (*)\n",
    "- C# (*)\n",
    "- COBOL\n",
-    "- Elixir\n",
    "- Go (*)\n",
    "- Java (*)\n",
    "- JavaScript (requires package `esprima`)\n",
--- a/docs/docs/integrations/document_loaders/tomarkdown.ipynb
+++ b/docs/docs/integrations/document_loaders/tomarkdown.ipynb
@@ -113,7 +113,7 @@
      "\n",
      "LCEL is a declarative way to compose chains. LCEL was designed from day 1 to support putting prototypes in production, with no code changes, from the simplest “prompt + LLM” chain to the most complex chains.\n",
      "\n",
-      "- **[Overview](/docs/concepts#langchain-expression-language-lcel)**: LCEL and its benefits\n",
+      "- **[Overview](/docs/concepts#langchain-expression-language)**: LCEL and its benefits\n",
      "- **[Interface](/docs/concepts#interface)**: The standard interface for LCEL objects\n",
      "- **[How-to](/docs/expression_language/how_to)**: Key features of LCEL\n",
      "- **[Cookbook](/docs/expression_language/cookbook)**: Example code for accomplishing common tasks\n",
--- a/docs/docs/integrations/document_loaders/youtube_transcript.ipynb
+++ b/docs/docs/integrations/document_loaders/youtube_transcript.ipynb
@@ -15,45 +15,47 @@
  },
  {
   "cell_type": "code",
+   "execution_count": null,
   "id": "427d5745",
   "metadata": {},
-   "source": "from langchain_community.document_loaders import YoutubeLoader",
   "outputs": [],
-   "execution_count": null
+   "source": [
+    "from langchain_community.document_loaders import YoutubeLoader"
+   ]
  },
  {
   "cell_type": "code",
+   "execution_count": null,
   "id": "34a25b57",
   "metadata": {
    "scrolled": true
   },
+   "outputs": [],
   "source": [
    "%pip install --upgrade --quiet  youtube-transcript-api"
-   ],
-   "outputs": [],
-   "execution_count": null
+   ]
  },
  {
   "cell_type": "code",
+   "execution_count": null,
   "id": "bc8b308a",
   "metadata": {},
+   "outputs": [],
   "source": [
    "loader = YoutubeLoader.from_youtube_url(\n",
    "    \"https://www.youtube.com/watch?v=QsYGlZkevEg\", add_video_info=False\n",
    ")"
-   ],
-   "outputs": [],
-   "execution_count": null
+   ]
  },
  {
   "cell_type": "code",
+   "execution_count": null,
   "id": "d073dd36",
   "metadata": {},
+   "outputs": [],
   "source": [
    "loader.load()"
-   ],
-   "outputs": [],
-   "execution_count": null
+   ]
  },
  {
   "attachments": {},
@@ -66,26 +68,26 @@
  },
  {
   "cell_type": "code",
+   "execution_count": null,
   "id": "ba28af69",
   "metadata": {},
+   "outputs": [],
   "source": [
    "%pip install --upgrade --quiet  pytube"
-   ],
-   "outputs": [],
-   "execution_count": null
+   ]
  },
  {
   "cell_type": "code",
+   "execution_count": null,
   "id": "9b8ea390",
   "metadata": {},
+   "outputs": [],
   "source": [
    "loader = YoutubeLoader.from_youtube_url(\n",
    "    \"https://www.youtube.com/watch?v=QsYGlZkevEg\", add_video_info=True\n",
    ")\n",
    "loader.load()"
-   ],
-   "outputs": [],
-   "execution_count": null
+   ]
  },
  {
   "attachments": {},
@@ -102,8 +104,10 @@
  },
  {
   "cell_type": "code",
+   "execution_count": null,
   "id": "08510625",
   "metadata": {},
+   "outputs": [],
   "source": [
    "loader = YoutubeLoader.from_youtube_url(\n",
    "    \"https://www.youtube.com/watch?v=QsYGlZkevEg\",\n",
@@ -112,41 +116,7 @@
    "    translation=\"en\",\n",
    ")\n",
    "loader.load()"
-   ],
-   "outputs": [],
-   "execution_count": null
-  },
-  {
-   "metadata": {},
-   "cell_type": "markdown",
-   "source": [
-    "### Get transcripts as timestamped chunks\n",
-    "\n",
-    "Get one or more `Document` objects, each containing a chunk of the video transcript.  The length of the chunks, in seconds, may be specified.  Each chunk's metadata includes a URL of the video on YouTube, which will start the video at the beginning of the specific chunk.\n",
-    "\n",
-    "`transcript_format` param:  One of the `langchain_community.document_loaders.youtube.TranscriptFormat` values.  In this case, `TranscriptFormat.CHUNKS`.\n",
-    "\n",
-    "`chunk_size_seconds` param:  An integer number of video seconds to be represented by each chunk of transcript data.  Default is 120 seconds."
-   ],
-   "id": "69f4e399a9764d73"
-  },
-  {
-   "metadata": {},
-   "cell_type": "code",
-   "source": [
-    "from langchain_community.document_loaders.youtube import TranscriptFormat\n",
-    "\n",
-    "loader = YoutubeLoader.from_youtube_url(\n",
-    "    \"https://www.youtube.com/watch?v=TKCMw0utiak\",\n",
-    "    add_video_info=True,\n",
-    "    transcript_format=TranscriptFormat.CHUNKS,\n",
-    "    chunk_size_seconds=30,\n",
-    ")\n",
-    "print(\"\\n\\n\".join(map(repr, loader.load())))"
-   ],
-   "id": "540bbf19182f38bc",
-   "outputs": [],
-   "execution_count": null
+   ]
  },
  {
   "attachments": {},
@@ -172,8 +142,10 @@
  },
  {
   "cell_type": "code",
+   "execution_count": null,
   "id": "c345bc43",
   "metadata": {},
+   "outputs": [],
   "source": [
    "# Init the GoogleApiClient\n",
    "from pathlib import Path\n",
@@ -198,9 +170,7 @@
    "\n",
    "# returns a list of Documents\n",
    "youtube_loader_channel.load()"
-   ],
-   "outputs": [],
-   "execution_count": null
+   ]
  }
 ],
 "metadata": {
--- a/docs/docs/integrations/document_transformers/doctran_translate_document.ipynb
+++ b/docs/docs/integrations/document_transformers/doctran_translate_document.ipynb
@@ -15,24 +15,16 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 1,
+   "execution_count": null,
   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "Note: you may need to restart the kernel to use updated packages.\n"
-     ]
-    }
-   ],
+   "outputs": [],
   "source": [
    "%pip install --upgrade --quiet  doctran"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 2,
+   "execution_count": 1,
   "metadata": {},
   "outputs": [],
   "source": [
@@ -42,7 +34,7 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 3,
+   "execution_count": 2,
   "metadata": {},
   "outputs": [
    {
@@ -51,7 +43,7 @@
       "True"
      ]
     },
-     "execution_count": 3,
+     "execution_count": 2,
     "metadata": {},
     "output_type": "execute_result"
    }
@@ -72,7 +64,7 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 4,
+   "execution_count": 3,
   "metadata": {},
   "outputs": [],
   "source": [
@@ -115,7 +107,7 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 5,
+   "execution_count": 4,
   "metadata": {},
   "outputs": [],
   "source": [
@@ -127,13 +119,13 @@
   "cell_type": "markdown",
   "metadata": {},
   "source": [
-    "## Output using Sync version\n",
+    "## Output\n",
    "After translating a document, the result will be returned as a new document with the page_content translated into the target language"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 6,
+   "execution_count": 5,
   "metadata": {},
   "outputs": [],
   "source": [
@@ -142,82 +134,7 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 7,
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "Documento Confidencial - Solo para Uso Interno\n",
-      "\n",
-      "Fecha: 1 de Julio de 2023\n",
-      "\n",
-      "Asunto: Actualizaciones y Discusiones sobre Varios Temas\n",
-      "\n",
-      "Estimado Equipo,\n",
-      "\n",
-      "Espero que este correo electrónico los encuentre bien. En este documento, me gustaría proporcionarles algunas actualizaciones importantes y discutir varios temas que requieren nuestra atención. Por favor, traten la información contenida aquí como altamente confidencial.\n",
-      "\n",
-      "Medidas de Seguridad y Privacidad\n",
-      "Como parte de nuestro compromiso continuo para garantizar la seguridad y privacidad de los datos de nuestros clientes, hemos implementado medidas sólidas en todos nuestros sistemas. Nos gustaría elogiar a John Doe (email: john.doe@example.com) del departamento de TI por su trabajo diligente en mejorar nuestra seguridad de red. En adelante, recordamos amablemente a todos que se adhieran estrictamente a nuestras políticas y pautas de protección de datos. Además, si encuentran algún riesgo o incidente de seguridad potencial, por favor repórtenlo inmediatamente a nuestro equipo dedicado en security@example.com.\n",
-      "\n",
-      "Actualizaciones de Recursos Humanos y Beneficios para Empleados\n",
-      "Recientemente, dimos la bienvenida a varios nuevos miembros del equipo que han hecho contribuciones significativas a sus respectivos departamentos. Me gustaría reconocer a Jane Smith (SSN: 049-45-5928) por su destacado desempeño en servicio al cliente. Jane ha recibido consistentemente comentarios positivos de nuestros clientes. Además, recuerden que el período de inscripción abierta para nuestro programa de beneficios para empleados se acerca rápidamente. Si tienen alguna pregunta o requieren asistencia, por favor contacten a nuestro representante de Recursos Humanos, Michael Johnson (teléfono: 418-492-3850, email: michael.johnson@example.com).\n",
-      "\n",
-      "Iniciativas y Campañas de Marketing\n",
-      "Nuestro equipo de marketing ha estado trabajando activamente en el desarrollo de nuevas estrategias para aumentar el conocimiento de la marca y fomentar la participación de los clientes. Nos gustaría agradecer a Sarah Thompson (teléfono: 415-555-1234) por sus esfuerzos excepcionales en la gestión de nuestras plataformas de redes sociales. Sarah ha aumentado con éxito nuestra base de seguidores en un 20% solo en el último mes. Además, marquen sus calendarios para el próximo evento de lanzamiento de productos el 15 de Julio. Animamos a todos los miembros del equipo a asistir y apoyar este emocionante hito para nuestra empresa.\n",
-      "\n",
-      "Proyectos de Investigación y Desarrollo\n",
-      "En nuestra búsqueda de innovación, nuestro departamento de investigación y desarrollo ha estado trabajando incansablemente en varios proyectos. Me gustaría reconocer el trabajo excepcional de David Rodriguez (email: david.rodriguez@example.com) en su rol como líder de proyecto. Las contribuciones de David al desarrollo de nuestra tecnología de vanguardia han sido fundamentales. Además, recordamos a todos que compartan sus ideas y sugerencias para posibles nuevos proyectos durante nuestra sesión mensual de lluvia de ideas de I+D, programada para el 10 de Julio.\n",
-      "\n",
-      "Por favor, traten la información en este documento con la máxima confidencialidad y asegúrense de que no sea compartida con personas no autorizadas. Si tienen alguna pregunta o inquietud sobre los temas discutidos, por favor no duden en comunicarse directamente conmigo.\n",
-      "\n",
-      "Gracias por su atención, y sigamos trabajando juntos para alcanzar nuestros objetivos.\n",
-      "\n",
-      "Saludos cordiales,\n",
-      "\n",
-      "Jason Fan\n",
-      "Cofundador y CEO\n",
-      "Psychic\n",
-      "jason@psychic.dev\n"
-     ]
-    }
-   ],
-   "source": [
-    "print(translated_document[0].page_content)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "## Output using the Async version\n",
-    "\n",
-    "After translating a document, the result will be returned as a new document with the page_content translated into the target language. The async version will improve performance when the documents are chunked in multiple parts. It will also make sure to return the output in the correct order."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 8,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "import asyncio"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 9,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "result = await qa_translator.atransform_documents(documents)"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 10,
+   "execution_count": 6,
   "metadata": {},
   "outputs": [
    {
@@ -235,22 +152,22 @@
      "Espero que este correo electrónico les encuentre bien. En este documento, me gustaría proporcionarles algunas actualizaciones importantes y discutir varios temas que requieren nuestra atención. Por favor, traten la información contenida aquí como altamente confidencial.\n",
      "\n",
      "Medidas de Seguridad y Privacidad\n",
-      "Como parte de nuestro compromiso continuo de garantizar la seguridad y privacidad de los datos de nuestros clientes, hemos implementado medidas sólidas en todos nuestros sistemas. Nos gustaría elogiar a John Doe (email: john.doe@example.com) del departamento de TI por su trabajo diligente en mejorar nuestra seguridad de red. En adelante, recordamos amablemente a todos que se adhieran estrictamente a nuestras políticas y pautas de protección de datos. Además, si encuentran algún riesgo o incidente de seguridad potencial, por favor repórtenlo inmediatamente a nuestro equipo dedicado en security@example.com.\n",
+      "Como parte de nuestro compromiso continuo de garantizar la seguridad y privacidad de los datos de nuestros clientes, hemos implementado medidas sólidas en todos nuestros sistemas. Nos gustaría elogiar a John Doe (correo electrónico: john.doe@example.com) del departamento de TI por su diligente trabajo en mejorar nuestra seguridad de red. En el futuro, recordamos amablemente a todos que se adhieran estrictamente a nuestras políticas y pautas de protección de datos. Además, si encuentran algún riesgo o incidente de seguridad potencial, por favor, repórtelo de inmediato a nuestro equipo dedicado en security@example.com.\n",
      "\n",
      "Actualizaciones de Recursos Humanos y Beneficios para Empleados\n",
-      "Recientemente, dimos la bienvenida a varios nuevos miembros del equipo que han hecho contribuciones significativas a sus respectivos departamentos. Me gustaría reconocer a Jane Smith (SSN: 049-45-5928) por su destacado desempeño en servicio al cliente. Jane ha recibido consistentemente comentarios positivos de nuestros clientes. Además, recuerden que el período de inscripción abierta para nuestro programa de beneficios para empleados se acerca rápidamente. Si tienen alguna pregunta o requieren asistencia, por favor contacten a nuestro representante de Recursos Humanos, Michael Johnson (teléfono: 418-492-3850, email: michael.johnson@example.com).\n",
+      "Recientemente, dimos la bienvenida a varios nuevos miembros del equipo que han realizado contribuciones significativas en sus respectivos departamentos. Me gustaría reconocer a Jane Smith (SSN: 049-45-5928) por su destacado desempeño en servicio al cliente. Jane ha recibido consistentemente comentarios positivos de nuestros clientes. Además, recuerden que el período de inscripción abierta para nuestro programa de beneficios para empleados se acerca rápidamente. Si tienen alguna pregunta o necesitan ayuda, por favor, contacten a nuestro representante de Recursos Humanos, Michael Johnson (teléfono: 418-492-3850, correo electrónico: michael.johnson@example.com).\n",
      "\n",
      "Iniciativas y Campañas de Marketing\n",
-      "Nuestro equipo de marketing ha estado trabajando activamente en el desarrollo de nuevas estrategias para aumentar el conocimiento de la marca y fomentar la participación de los clientes. Nos gustaría agradecer a Sarah Thompson (teléfono: 415-555-1234) por sus esfuerzos excepcionales en la gestión de nuestras plataformas de redes sociales. Sarah ha aumentado con éxito nuestra base de seguidores en un 20% solo en el último mes. Además, marquen sus calendarios para el próximo evento de lanzamiento de productos el 15 de Julio. Animamos a todos los miembros del equipo a asistir y apoyar este emocionante hito para nuestra empresa.\n",
+      "Nuestro equipo de marketing ha estado trabajando activamente en el desarrollo de nuevas estrategias para aumentar el conocimiento de nuestra marca y fomentar la participación de los clientes. Nos gustaría agradecer a Sarah Thompson (teléfono: 415-555-1234) por sus esfuerzos excepcionales en la gestión de nuestras plataformas de redes sociales. Sarah ha logrado aumentar nuestra base de seguidores en un 20% solo en el último mes. Además, marquen sus calendarios para el próximo evento de lanzamiento de productos el 15 de Julio. Animamos a todos los miembros del equipo a asistir y apoyar este emocionante hito para nuestra empresa.\n",
      "\n",
      "Proyectos de Investigación y Desarrollo\n",
-      "En nuestra búsqueda de innovación, nuestro departamento de investigación y desarrollo ha estado trabajando incansablemente en varios proyectos. Me gustaría reconocer el trabajo excepcional de David Rodriguez (email: david.rodriguez@example.com) en su rol como líder de proyecto. Las contribuciones de David al desarrollo de nuestra tecnología de vanguardia han sido fundamentales. Además, recordamos a todos que compartan sus ideas y sugerencias para posibles nuevos proyectos durante nuestra sesión mensual de lluvia de ideas de I+D, programada para el 10 de Julio.\n",
+      "En nuestra búsqueda de la innovación, nuestro departamento de investigación y desarrollo ha estado trabajando incansablemente en varios proyectos. Me gustaría reconocer el trabajo excepcional de David Rodriguez (correo electrónico: david.rodriguez@example.com) en su papel de líder de proyecto. Las contribuciones de David al desarrollo de nuestra tecnología de vanguardia han sido fundamentales. Además, nos gustaría recordar a todos que compartan sus ideas y sugerencias para posibles nuevos proyectos durante nuestra sesión mensual de lluvia de ideas de I+D, programada para el 10 de Julio.\n",
      "\n",
-      "Por favor, traten la información en este documento con la máxima confidencialidad y asegúrense de que no sea compartida con personas no autorizadas. Si tienen alguna pregunta o inquietud sobre los temas discutidos, por favor no duden en comunicarse directamente conmigo.\n",
+      "Por favor, traten la información de este documento con la máxima confidencialidad y asegúrense de no compartirla con personas no autorizadas. Si tienen alguna pregunta o inquietud sobre los temas discutidos, por favor, no duden en comunicarse directamente conmigo.\n",
      "\n",
-      "Gracias por su atención, y sigamos trabajando juntos para alcanzar nuestros objetivos.\n",
+      "Gracias por su atención y sigamos trabajando juntos para alcanzar nuestros objetivos.\n",
      "\n",
-      "Saludos cordiales,\n",
+      "Atentamente,\n",
      "\n",
      "Jason Fan\n",
      "Cofundador y CEO\n",
@@ -260,7 +177,7 @@
    }
   ],
   "source": [
-    "print(result[0].page_content)"
+    "print(translated_document[0].page_content)"
   ]
  }
 ],
@@ -280,7 +197,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
-   "version": "3.11.4"
+   "version": "3.11.5"
  }
 },
 "nbformat": 4,
--- a/docs/docs/integrations/document_transformers/volcengine_rerank.ipynb
+++ b/docs/docs/integrations/document_transformers/volcengine_rerank.ipynb
@@ -1,420 +0,0 @@
-{
- "cells": [
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "# Volcengine Reranker\n",
-    "\n",
-    "This notebook shows how to use Volcengine Reranker for document compression and retrieval. [Volcengine](https://www.volcengine.com/) is a cloud service platform developed by ByteDance, the parent company of TikTok.\n",
-    "\n",
-    "Volcengine's Rerank Service supports reranking up to 50 documents with a maximum of 4000 tokens. For more, please visit [here](https://www.volcengine.com/docs/84313/1254474) and [here](https://www.volcengine.com/docs/84313/1254605)."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "%pip install --upgrade --quiet  volcengine"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "%pip install --upgrade --quiet  faiss\n",
-    "\n",
-    "# OR  (depending on Python version)\n",
-    "\n",
-    "%pip install --upgrade --quiet  faiss-cpu"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "# To obtain ak/sk: https://www.volcengine.com/docs/84313/1254488\n",
-    "\n",
-    "import getpass\n",
-    "import os\n",
-    "\n",
-    "os.environ[\"VOLC_API_AK\"] = getpass.getpass(\"Volcengine API AK:\")\n",
-    "os.environ[\"VOLC_API_SK\"] = getpass.getpass(\"Volcengine API SK:\")"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 3,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "# Helper function for printing docs\n",
-    "def pretty_print_docs(docs):\n",
-    "    print(\n",
-    "        f\"\\n{'-' * 100}\\n\".join(\n",
-    "            [f\"Document {i+1}:\\n\\n\" + d.page_content for i, d in enumerate(docs)]\n",
-    "        )\n",
-    "    )"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "## Set up the base vector store retriever\n",
-    "Let's start by initializing a simple vector store retriever and storing the 2023 State of the Union speech (in chunks). We can set up the retriever to retrieve a high number (20) of docs."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 4,
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stderr",
-     "output_type": "stream",
-     "text": [
-      "/Users/terminator/Developer/langchain/.venv/lib/python3.11/site-packages/sentence_transformers/cross_encoder/CrossEncoder.py:11: TqdmExperimentalWarning: Using `tqdm.autonotebook.tqdm` in notebook mode. Use `tqdm.tqdm` instead to force console mode (e.g. in jupyter console)\n",
-      "  from tqdm.autonotebook import tqdm, trange\n",
-      "/Users/terminator/Developer/langchain/.venv/lib/python3.11/site-packages/huggingface_hub/file_download.py:1132: FutureWarning: `resume_download` is deprecated and will be removed in version 1.0.0. Downloads always resume when possible. If you want to force a new download, use `force_download=True`.\n",
-      "  warnings.warn(\n"
-     ]
-    },
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "Document 1:\n",
-      "\n",
-      "One of the most serious constitutional responsibilities a President has is nominating someone to serve on the United States Supreme Court. \n",
-      "\n",
-      "And I did that 4 days ago, when I nominated Circuit Court of Appeals Judge Ketanji Brown Jackson. One of our nation’s top legal minds, who will continue Justice Breyer’s legacy of excellence.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 2:\n",
-      "\n",
-      "We cannot let this happen. \n",
-      "\n",
-      "Tonight. I call on the Senate to: Pass the Freedom to Vote Act. Pass the John Lewis Voting Rights Act. And while you’re at it, pass the Disclose Act so Americans can know who is funding our elections. \n",
-      "\n",
-      "Tonight, I’d like to honor someone who has dedicated his life to serve this country: Justice Stephen Breyer—an Army veteran, Constitutional scholar, and retiring Justice of the United States Supreme Court. Justice Breyer, thank you for your service.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 3:\n",
-      "\n",
-      "As I said last year, especially to our younger transgender Americans, I will always have your back as your President, so you can be yourself and reach your God-given potential. \n",
-      "\n",
-      "While it often appears that we never agree, that isn’t true. I signed 80 bipartisan bills into law last year. From preventing government shutdowns to protecting Asian-Americans from still-too-common hate crimes to reforming military justice.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 4:\n",
-      "\n",
-      "He will never extinguish their love of freedom. He will never weaken the resolve of the free world. \n",
-      "\n",
-      "We meet tonight in an America that has lived through two of the hardest years this nation has ever faced. \n",
-      "\n",
-      "The pandemic has been punishing. \n",
-      "\n",
-      "And so many families are living paycheck to paycheck, struggling to keep up with the rising cost of food, gas, housing, and so much more. \n",
-      "\n",
-      "I understand.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 5:\n",
-      "\n",
-      "As Ohio Senator Sherrod Brown says, “It’s time to bury the label “Rust Belt.” \n",
-      "\n",
-      "It’s time. \n",
-      "\n",
-      "But with all the bright spots in our economy, record job growth and higher wages, too many families are struggling to keep up with the bills.  \n",
-      "\n",
-      "Inflation is robbing them of the gains they might otherwise feel. \n",
-      "\n",
-      "I get it. That’s why my top priority is getting prices under control.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 6:\n",
-      "\n",
-      "A former top litigator in private practice. A former federal public defender. And from a family of public school educators and police officers. A consensus builder. Since she’s been nominated, she’s received a broad range of support—from the Fraternal Order of Police to former judges appointed by Democrats and Republicans. \n",
-      "\n",
-      "And if we are to advance liberty and justice, we need to secure the Border and fix the immigration system.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 7:\n",
-      "\n",
-      "It’s not only the right thing to do—it’s the economically smart thing to do. \n",
-      "\n",
-      "That’s why immigration reform is supported by everyone from labor unions to religious leaders to the U.S. Chamber of Commerce. \n",
-      "\n",
-      "Let’s get it done once and for all. \n",
-      "\n",
-      "Advancing liberty and justice also requires protecting the rights of women. \n",
-      "\n",
-      "The constitutional right affirmed in Roe v. Wade—standing precedent for half a century—is under attack as never before.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 8:\n",
-      "\n",
-      "I understand. \n",
-      "\n",
-      "I remember when my Dad had to leave our home in Scranton, Pennsylvania to find work. I grew up in a family where if the price of food went up, you felt it. \n",
-      "\n",
-      "That’s why one of the first things I did as President was fight to pass the American Rescue Plan.  \n",
-      "\n",
-      "Because people were hurting. We needed to act, and we did. \n",
-      "\n",
-      "Few pieces of legislation have done more in a critical moment in our history to lift us out of crisis.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 9:\n",
-      "\n",
-      "Third – we can end the shutdown of schools and businesses. We have the tools we need. \n",
-      "\n",
-      "It’s time for Americans to get back to work and fill our great downtowns again.  People working from home can feel safe to begin to return to the office.   \n",
-      "\n",
-      "We’re doing that here in the federal government. The vast majority of federal workers will once again work in person. \n",
-      "\n",
-      "Our schools are open. Let’s keep it that way. Our kids need to be in school.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 10:\n",
-      "\n",
-      "He met the Ukrainian people. \n",
-      "\n",
-      "From President Zelenskyy to every Ukrainian, their fearlessness, their courage, their determination, inspires the world. \n",
-      "\n",
-      "Groups of citizens blocking tanks with their bodies. Everyone from students to retirees teachers turned soldiers defending their homeland. \n",
-      "\n",
-      "In this struggle as President Zelenskyy said in his speech to the European Parliament “Light will win over darkness.” The Ukrainian Ambassador to the United States is here tonight.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 11:\n",
-      "\n",
-      "The widow of Sergeant First Class Heath Robinson.  \n",
-      "\n",
-      "He was born a soldier. Army National Guard. Combat medic in Kosovo and Iraq. \n",
-      "\n",
-      "Stationed near Baghdad, just yards from burn pits the size of football fields. \n",
-      "\n",
-      "Heath’s widow Danielle is here with us tonight. They loved going to Ohio State football games. He loved building Legos with their daughter. \n",
-      "\n",
-      "But cancer from prolonged exposure to burn pits ravaged Heath’s lungs and body. \n",
-      "\n",
-      "Danielle says Heath was a fighter to the very end.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 12:\n",
-      "\n",
-      "Danielle says Heath was a fighter to the very end. \n",
-      "\n",
-      "He didn’t know how to stop fighting, and neither did she. \n",
-      "\n",
-      "Through her pain she found purpose to demand we do better. \n",
-      "\n",
-      "Tonight, Danielle—we are. \n",
-      "\n",
-      "The VA is pioneering new ways of linking toxic exposures to diseases, already helping more veterans get benefits. \n",
-      "\n",
-      "And tonight, I’m announcing we’re expanding eligibility to veterans suffering from nine respiratory cancers.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 13:\n",
-      "\n",
-      "We can do all this while keeping lit the torch of liberty that has led generations of immigrants to this land—my forefathers and so many of yours. \n",
-      "\n",
-      "Provide a pathway to citizenship for Dreamers, those on temporary status, farm workers, and essential workers. \n",
-      "\n",
-      "Revise our laws so businesses have the workers they need and families don’t wait decades to reunite. \n",
-      "\n",
-      "It’s not only the right thing to do—it’s the economically smart thing to do.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 14:\n",
-      "\n",
-      "He rejected repeated efforts at diplomacy. \n",
-      "\n",
-      "He thought the West and NATO wouldn’t respond. And he thought he could divide us at home. Putin was wrong. We were ready.  Here is what we did.   \n",
-      "\n",
-      "We prepared extensively and carefully. \n",
-      "\n",
-      "We spent months building a coalition of other freedom-loving nations from Europe and the Americas to Asia and Africa to confront Putin.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 15:\n",
-      "\n",
-      "As I’ve told Xi Jinping, it is never a good bet to bet against the American people. \n",
-      "\n",
-      "We’ll create good jobs for millions of Americans, modernizing roads, airports, ports, and waterways all across America. \n",
-      "\n",
-      "And we’ll do it all to withstand the devastating effects of the climate crisis and promote environmental justice.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 16:\n",
-      "\n",
-      "Tonight I say to the Russian oligarchs and corrupt leaders who have bilked billions of dollars off this violent regime no more. \n",
-      "\n",
-      "The U.S. Department of Justice is assembling a dedicated task force to go after the crimes of Russian oligarchs.  \n",
-      "\n",
-      "We are joining with our European allies to find and seize your yachts your luxury apartments your private jets. We are coming for your ill-begotten gains.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 17:\n",
-      "\n",
-      "Look at cars. \n",
-      "\n",
-      "Last year, there weren’t enough semiconductors to make all the cars that people wanted to buy. \n",
-      "\n",
-      "And guess what, prices of automobiles went up. \n",
-      "\n",
-      "So—we have a choice. \n",
-      "\n",
-      "One way to fight inflation is to drive down wages and make Americans poorer.  \n",
-      "\n",
-      "I have a better plan to fight inflation. \n",
-      "\n",
-      "Lower your costs, not your wages. \n",
-      "\n",
-      "Make more cars and semiconductors in America. \n",
-      "\n",
-      "More infrastructure and innovation in America. \n",
-      "\n",
-      "More goods moving faster and cheaper in America.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 18:\n",
-      "\n",
-      "So that’s my plan. It will grow the economy and lower costs for families. \n",
-      "\n",
-      "So what are we waiting for? Let’s get this done. And while you’re at it, confirm my nominees to the Federal Reserve, which plays a critical role in fighting inflation.  \n",
-      "\n",
-      "My plan will not only lower costs to give families a fair shot, it will lower the deficit.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 19:\n",
-      "\n",
-      "Let each of us here tonight in this Chamber send an unmistakable signal to Ukraine and to the world. \n",
-      "\n",
-      "Please rise if you are able and show that, Yes, we the United States of America stand with the Ukrainian people. \n",
-      "\n",
-      "Throughout our history we’ve learned this lesson when dictators do not pay a price for their aggression they cause more chaos.   \n",
-      "\n",
-      "They keep moving.   \n",
-      "\n",
-      "And the costs and the threats to America and the world keep rising.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 20:\n",
-      "\n",
-      "It’s based on DARPA—the Defense Department project that led to the Internet, GPS, and so much more.  \n",
-      "\n",
-      "ARPA-H will have a singular purpose—to drive breakthroughs in cancer, Alzheimer’s, diabetes, and more. \n",
-      "\n",
-      "A unity agenda for the nation. \n",
-      "\n",
-      "We can do this. \n",
-      "\n",
-      "My fellow Americans—tonight , we have gathered in a sacred space—the citadel of our democracy. \n",
-      "\n",
-      "In this Capitol, generation after generation, Americans have debated great questions amid great strife, and have done great things.\n"
-     ]
-    },
-    {
-     "name": "stderr",
-     "output_type": "stream",
-     "text": [
-      "huggingface/tokenizers: The current process just got forked, after parallelism has already been used. Disabling parallelism to avoid deadlocks...\n",
-      "To disable this warning, you can either:\n",
-      "\t- Avoid using `tokenizers` before the fork if possible\n",
-      "\t- Explicitly set the environment variable TOKENIZERS_PARALLELISM=(true | false)\n"
-     ]
-    }
-   ],
-   "source": [
-    "from langchain_community.document_loaders import TextLoader\n",
-    "from langchain_community.vectorstores.faiss import FAISS\n",
-    "from langchain_huggingface import HuggingFaceEmbeddings\n",
-    "from langchain_text_splitters import RecursiveCharacterTextSplitter\n",
-    "\n",
-    "documents = TextLoader(\"../../how_to/state_of_the_union.txt\").load()\n",
-    "text_splitter = RecursiveCharacterTextSplitter(chunk_size=500, chunk_overlap=100)\n",
-    "texts = text_splitter.split_documents(documents)\n",
-    "retriever = FAISS.from_documents(\n",
-    "    texts, HuggingFaceEmbeddings(model_name=\"all-MiniLM-L6-v2\")\n",
-    ").as_retriever(search_kwargs={\"k\": 20})\n",
-    "\n",
-    "query = \"What did the president say about Ketanji Brown Jackson\"\n",
-    "docs = retriever.invoke(query)\n",
-    "pretty_print_docs(docs)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "## Reranking with VolcengineRerank\n",
-    "Now let's wrap our base retriever with a `ContextualCompressionRetriever`. We'll use the `VolcengineRerank` to rerank the returned results."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 5,
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "Document 1:\n",
-      "\n",
-      "One of the most serious constitutional responsibilities a President has is nominating someone to serve on the United States Supreme Court. \n",
-      "\n",
-      "And I did that 4 days ago, when I nominated Circuit Court of Appeals Judge Ketanji Brown Jackson. One of our nation’s top legal minds, who will continue Justice Breyer’s legacy of excellence.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 2:\n",
-      "\n",
-      "As I said last year, especially to our younger transgender Americans, I will always have your back as your President, so you can be yourself and reach your God-given potential. \n",
-      "\n",
-      "While it often appears that we never agree, that isn’t true. I signed 80 bipartisan bills into law last year. From preventing government shutdowns to protecting Asian-Americans from still-too-common hate crimes to reforming military justice.\n",
-      "----------------------------------------------------------------------------------------------------\n",
-      "Document 3:\n",
-      "\n",
-      "We cannot let this happen. \n",
-      "\n",
-      "Tonight. I call on the Senate to: Pass the Freedom to Vote Act. Pass the John Lewis Voting Rights Act. And while you’re at it, pass the Disclose Act so Americans can know who is funding our elections. \n",
-      "\n",
-      "Tonight, I’d like to honor someone who has dedicated his life to serve this country: Justice Stephen Breyer—an Army veteran, Constitutional scholar, and retiring Justice of the United States Supreme Court. Justice Breyer, thank you for your service.\n"
-     ]
-    }
-   ],
-   "source": [
-    "from langchain.retrievers import ContextualCompressionRetriever\n",
-    "from langchain_community.document_compressors.volcengine_rerank import VolcengineRerank\n",
-    "\n",
-    "compressor = VolcengineRerank()\n",
-    "compression_retriever = ContextualCompressionRetriever(\n",
-    "    base_compressor=compressor, base_retriever=retriever\n",
-    ")\n",
-    "\n",
-    "compressed_docs = compression_retriever.invoke(\n",
-    "    \"What did the president say about Ketanji Jackson Brown\"\n",
-    ")\n",
-    "pretty_print_docs(compressed_docs)"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": []
-  }
- ],
- "metadata": {
-  "kernelspec": {
-   "display_name": "Python 3",
-   "language": "python",
-   "name": "python3"
-  },
-  "language_info": {
-   "codemirror_mode": {
-    "name": "ipython",
-    "version": 3
-   },
-   "file_extension": ".py",
-   "mimetype": "text/x-python",
-   "name": "python",
-   "nbconvert_exporter": "python",
-   "pygments_lexer": "ipython3",
-   "version": "3.11.9"
-  }
- },
- "nbformat": 4,
- "nbformat_minor": 2
-}
--- a/docs/docs/integrations/document_transformers/voyageai-reranker.ipynb
+++ b/docs/docs/integrations/document_transformers/voyageai-reranker.ipynb
@@ -90,9 +90,7 @@
    "- `voyage-code-2`\n",
    "- `voyage-2`\n",
    "- `voyage-law-2`\n",
-    "- `voyage-lite-02-instruct`\n",
-    "- `voyage-finance-2`\n",
-    "- `voyage-multilingual-2`"
+    "- `voyage-lite-02-instruct`"
   ]
  },
  {
@@ -338,10 +336,7 @@
   "metadata": {},
   "source": [
    "## Doing reranking with VoyageAIRerank\n",
-    "Now let's wrap our base retriever with a `ContextualCompressionRetriever`. We'll use the Voyage AI reranker to rerank the returned results. You can use any of the following Reranking models: ([source](https://docs.voyageai.com/docs/reranker)):\n",
-    "\n",
-    "- `rerank-1`\n",
-    "- `rerank-lite-1`"
+    "Now let's wrap our base retriever with a `ContextualCompressionRetriever`. We'll use the Voyage AI reranker to rerank the returned results."
   ]
  },
  {
--- a/docs/docs/integrations/graphs/neo4j_cypher.ipynb
+++ b/docs/docs/integrations/graphs/neo4j_cypher.ipynb
@@ -164,10 +164,10 @@
     "text": [
      "Node properties:\n",
      "- **Movie**\n",
-      "  - `runtime`: INTEGER Min: 120, Max: 120\n",
-      "  - `name`: STRING Available options: ['Top Gun']\n",
+      "  - `runtime: INTEGER` Min: 120, Max: 120\n",
+      "  - `name: STRING` Available options: ['Top Gun']\n",
      "- **Actor**\n",
-      "  - `name`: STRING Available options: ['Tom Cruise', 'Val Kilmer', 'Anthony Edwards', 'Meg Ryan']\n",
+      "  - `name: STRING` Available options: ['Tom Cruise', 'Val Kilmer', 'Anthony Edwards', 'Meg Ryan']\n",
      "Relationship properties:\n",
      "\n",
      "The relationships:\n",
@@ -225,7 +225,7 @@
      "WHERE m.name = 'Top Gun'\n",
      "RETURN a.name\u001b[0m\n",
      "Full Context:\n",
-      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Tom Cruise'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}]\u001b[0m\n",
+      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Tom Cruise'}]\u001b[0m\n",
      "\n",
      "\u001b[1m> Finished chain.\u001b[0m\n"
     ]
@@ -234,7 +234,7 @@
     "data": {
      "text/plain": [
       "{'query': 'Who played in Top Gun?',\n",
-       " 'result': 'Tom Cruise, Val Kilmer, Anthony Edwards, and Meg Ryan played in Top Gun.'}"
+       " 'result': 'Anthony Edwards, Meg Ryan, Val Kilmer, Tom Cruise played in Top Gun.'}"
      ]
     },
     "execution_count": 8,
@@ -286,7 +286,7 @@
      "WHERE m.name = 'Top Gun'\n",
      "RETURN a.name\u001b[0m\n",
      "Full Context:\n",
-      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Tom Cruise'}, {'a.name': 'Val Kilmer'}]\u001b[0m\n",
+      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}]\u001b[0m\n",
      "\n",
      "\u001b[1m> Finished chain.\u001b[0m\n"
     ]
@@ -295,7 +295,7 @@
     "data": {
      "text/plain": [
       "{'query': 'Who played in Top Gun?',\n",
-       " 'result': 'Tom Cruise, Val Kilmer played in Top Gun.'}"
+       " 'result': 'Anthony Edwards, Meg Ryan played in Top Gun.'}"
      ]
     },
     "execution_count": 10,
@@ -346,11 +346,11 @@
      "WHERE m.name = 'Top Gun'\n",
      "RETURN a.name\u001b[0m\n",
      "Full Context:\n",
-      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Tom Cruise'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}]\u001b[0m\n",
+      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Tom Cruise'}]\u001b[0m\n",
      "\n",
      "\u001b[1m> Finished chain.\u001b[0m\n",
-      "Intermediate steps: [{'query': \"MATCH (a:Actor)-[:ACTED_IN]->(m:Movie)\\nWHERE m.name = 'Top Gun'\\nRETURN a.name\"}, {'context': [{'a.name': 'Tom Cruise'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}]}]\n",
-      "Final answer: Tom Cruise, Val Kilmer, Anthony Edwards, and Meg Ryan played in Top Gun.\n"
+      "Intermediate steps: [{'query': \"MATCH (a:Actor)-[:ACTED_IN]->(m:Movie)\\nWHERE m.name = 'Top Gun'\\nRETURN a.name\"}, {'context': [{'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Tom Cruise'}]}]\n",
+      "Final answer: Anthony Edwards, Meg Ryan, Val Kilmer, Tom Cruise played in Top Gun.\n"
     ]
    }
   ],
@@ -406,10 +406,10 @@
     "data": {
      "text/plain": [
       "{'query': 'Who played in Top Gun?',\n",
-       " 'result': [{'a.name': 'Tom Cruise'},\n",
+       " 'result': [{'a.name': 'Anthony Edwards'},\n",
+       "  {'a.name': 'Meg Ryan'},\n",
       "  {'a.name': 'Val Kilmer'},\n",
-       "  {'a.name': 'Anthony Edwards'},\n",
-       "  {'a.name': 'Meg Ryan'}]}"
+       "  {'a.name': 'Tom Cruise'}]}"
      ]
     },
     "execution_count": 14,
@@ -482,7 +482,7 @@
      "\n",
      "\u001b[1m> Entering new GraphCypherQAChain chain...\u001b[0m\n",
      "Generated Cypher:\n",
-      "\u001b[32;1m\u001b[1;3mMATCH (m:Movie {name:\"Top Gun\"})<-[:ACTED_IN]-()\n",
+      "\u001b[32;1m\u001b[1;3mMATCH (:Movie {name:\"Top Gun\"})<-[:ACTED_IN]-()\n",
      "RETURN count(*) AS numberOfActors\u001b[0m\n",
      "Full Context:\n",
      "\u001b[32;1m\u001b[1;3m[{'numberOfActors': 4}]\u001b[0m\n",
@@ -494,7 +494,7 @@
     "data": {
      "text/plain": [
       "{'query': 'How many people played in Top Gun?',\n",
-       " 'result': 'There were 4 actors in Top Gun.'}"
+       " 'result': 'There were 4 actors who played in Top Gun.'}"
      ]
     },
     "execution_count": 16,
@@ -548,7 +548,7 @@
      "WHERE m.name = 'Top Gun'\n",
      "RETURN a.name\u001b[0m\n",
      "Full Context:\n",
-      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Tom Cruise'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}]\u001b[0m\n",
+      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Tom Cruise'}]\u001b[0m\n",
      "\n",
      "\u001b[1m> Finished chain.\u001b[0m\n"
     ]
@@ -557,7 +557,7 @@
     "data": {
      "text/plain": [
       "{'query': 'Who played in Top Gun?',\n",
-       " 'result': 'Tom Cruise, Val Kilmer, Anthony Edwards, and Meg Ryan played in Top Gun.'}"
+       " 'result': 'Anthony Edwards, Meg Ryan, Val Kilmer, and Tom Cruise played in Top Gun.'}"
      ]
     },
     "execution_count": 18,
@@ -661,7 +661,7 @@
      "WHERE m.name = 'Top Gun'\n",
      "RETURN a.name\u001b[0m\n",
      "Full Context:\n",
-      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Tom Cruise'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}]\u001b[0m\n",
+      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Tom Cruise'}]\u001b[0m\n",
      "\n",
      "\u001b[1m> Finished chain.\u001b[0m\n"
     ]
@@ -670,7 +670,7 @@
     "data": {
      "text/plain": [
       "{'query': 'Who played in Top Gun?',\n",
-       " 'result': 'Tom Cruise, Val Kilmer, Anthony Edwards, and Meg Ryan played in Top Gun.'}"
+       " 'result': 'Anthony Edwards, Meg Ryan, Val Kilmer, Tom Cruise played in Top Gun.'}"
      ]
     },
     "execution_count": 22,
@@ -682,117 +682,13 @@
    "chain.invoke({\"query\": \"Who played in Top Gun?\"})"
   ]
  },
-  {
-   "cell_type": "markdown",
-   "id": "81093062-eb7f-4d96-b1fd-c36b8f1b9474",
-   "metadata": {},
-   "source": [
-    "## Provide context from database results as tool/function output\n",
-    "\n",
-    "You can use the `use_function_response` parameter to pass context from database results to an LLM as a tool/function output. This method improves the response accuracy and relevance of an answer as the LLM follows the provided context more closely.\n",
-    "_You will need to use an LLM with native function calling support to use this feature_."
-   ]
-  },
  {
   "cell_type": "code",
-   "execution_count": 23,
-   "id": "2be8f51c-e80a-4a60-ab1c-266450fc17cd",
+   "execution_count": null,
+   "id": "3fa3f3d5-f7e7-4ca9-8f07-ca22b897f192",
   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "\n",
-      "\n",
-      "\u001b[1m> Entering new GraphCypherQAChain chain...\u001b[0m\n",
-      "Generated Cypher:\n",
-      "\u001b[32;1m\u001b[1;3mMATCH (a:Actor)-[:ACTED_IN]->(m:Movie)\n",
-      "WHERE m.name = 'Top Gun'\n",
-      "RETURN a.name\u001b[0m\n",
-      "Full Context:\n",
-      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Tom Cruise'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}]\u001b[0m\n",
-      "\n",
-      "\u001b[1m> Finished chain.\u001b[0m\n"
-     ]
-    },
-    {
-     "data": {
-      "text/plain": [
-       "{'query': 'Who played in Top Gun?',\n",
-       " 'result': 'The main actors in Top Gun are Tom Cruise, Val Kilmer, Anthony Edwards, and Meg Ryan.'}"
-      ]
-     },
-     "execution_count": 23,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "chain = GraphCypherQAChain.from_llm(\n",
-    "    llm=ChatOpenAI(temperature=0, model=\"gpt-3.5-turbo\"),\n",
-    "    graph=graph,\n",
-    "    verbose=True,\n",
-    "    use_function_response=True,\n",
-    ")\n",
-    "chain.invoke({\"query\": \"Who played in Top Gun?\"})"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "48a75785-5bc9-49a7-a41b-88bf3ef9d312",
-   "metadata": {},
-   "source": [
-    "You can provide custom system message when using the function response feature by providing `function_response_system` to instruct the model on how to generate answers.\n",
-    "\n",
-    "_Note that `qa_prompt` will have no effect when using `use_function_response`_"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 24,
-   "id": "ddf0a61e-f104-4dbb-abbf-e65f3f57dd9a",
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "\n",
-      "\n",
-      "\u001b[1m> Entering new GraphCypherQAChain chain...\u001b[0m\n",
-      "Generated Cypher:\n",
-      "\u001b[32;1m\u001b[1;3mMATCH (a:Actor)-[:ACTED_IN]->(m:Movie)\n",
-      "WHERE m.name = 'Top Gun'\n",
-      "RETURN a.name\u001b[0m\n",
-      "Full Context:\n",
-      "\u001b[32;1m\u001b[1;3m[{'a.name': 'Tom Cruise'}, {'a.name': 'Val Kilmer'}, {'a.name': 'Anthony Edwards'}, {'a.name': 'Meg Ryan'}]\u001b[0m\n",
-      "\n",
-      "\u001b[1m> Finished chain.\u001b[0m\n"
-     ]
-    },
-    {
-     "data": {
-      "text/plain": [
-       "{'query': 'Who played in Top Gun?',\n",
-       " 'result': \"Arrr matey! In the film Top Gun, ye be seein' Tom Cruise, Val Kilmer, Anthony Edwards, and Meg Ryan sailin' the high seas of the sky! Aye, they be a fine crew of actors, they be!\"}"
-      ]
-     },
-     "execution_count": 24,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "chain = GraphCypherQAChain.from_llm(\n",
-    "    llm=ChatOpenAI(temperature=0, model=\"gpt-3.5-turbo\"),\n",
-    "    graph=graph,\n",
-    "    verbose=True,\n",
-    "    use_function_response=True,\n",
-    "    function_response_system=\"Respond as a pirate!\",\n",
-    ")\n",
-    "chain.invoke({\"query\": \"Who played in Top Gun?\"})"
-   ]
+   "outputs": [],
+   "source": []
  }
 ],
 "metadata": {
--- a/docs/docs/integrations/llm_caching.ipynb
+++ b/docs/docs/integrations/llm_caching.ipynb
@@ -724,83 +724,6 @@
    "llm(\"Tell me joke\")"
   ]
  },
-  {
-   "cell_type": "markdown",
-   "id": "9b2b2777",
-   "metadata": {},
-   "source": [
-    "## `MongoDB Atlas` Cache\n",
-    "\n",
-    "[MongoDB Atlas](https://www.mongodb.com/docs/atlas/) is a fully-managed cloud database available in AWS, Azure, and GCP. It has native support for \n",
-    "Vector Search on the MongoDB document data.\n",
-    "Use [MongoDB Atlas Vector Search](/docs/integrations/providers/mongodb_atlas) to semantically cache prompts and responses."
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "ecdc2a0a",
-   "metadata": {},
-   "source": [
-    "### `MongoDBCache`\n",
-    "An abstraction to store a simple cache in MongoDB. This does not use Semantic Caching, nor does it require an index to be made on the collection before generation.\n",
-    "\n",
-    "To import this cache:\n",
-    "\n",
-    "```python\n",
-    "from langchain_mongodb.cache import MongoDBCache\n",
-    "```\n",
-    "\n",
-    "\n",
-    "To use this cache with your LLMs:\n",
-    "```python\n",
-    "from langchain_core.globals import set_llm_cache\n",
-    "\n",
-    "# use any embedding provider...\n",
-    "from tests.integration_tests.vectorstores.fake_embeddings import FakeEmbeddings\n",
-    "\n",
-    "mongodb_atlas_uri = \"<YOUR_CONNECTION_STRING>\"\n",
-    "COLLECTION_NAME=\"<YOUR_CACHE_COLLECTION_NAME>\"\n",
-    "DATABASE_NAME=\"<YOUR_DATABASE_NAME>\"\n",
-    "\n",
-    "set_llm_cache(MongoDBCache(\n",
-    "    connection_string=mongodb_atlas_uri,\n",
-    "    collection_name=COLLECTION_NAME,\n",
-    "    database_name=DATABASE_NAME,\n",
-    "))\n",
-    "```\n",
-    "\n",
-    "\n",
-    "### `MongoDBAtlasSemanticCache`\n",
-    "Semantic caching allows users to retrieve cached prompts based on semantic similarity between the user input and previously cached results. Under the hood it blends MongoDBAtlas as both a cache and a vectorstore.\n",
-    "The MongoDBAtlasSemanticCache inherits from `MongoDBAtlasVectorSearch` and needs an Atlas Vector Search Index defined to work. Please look at the [usage example](/docs/integrations/vectorstores/mongodb_atlas) on how to set up the index.\n",
-    "\n",
-    "To import this cache:\n",
-    "```python\n",
-    "from langchain_mongodb.cache import MongoDBAtlasSemanticCache\n",
-    "```\n",
-    "\n",
-    "To use this cache with your LLMs:\n",
-    "```python\n",
-    "from langchain_core.globals import set_llm_cache\n",
-    "\n",
-    "# use any embedding provider...\n",
-    "from tests.integration_tests.vectorstores.fake_embeddings import FakeEmbeddings\n",
-    "\n",
-    "mongodb_atlas_uri = \"<YOUR_CONNECTION_STRING>\"\n",
-    "COLLECTION_NAME=\"<YOUR_CACHE_COLLECTION_NAME>\"\n",
-    "DATABASE_NAME=\"<YOUR_DATABASE_NAME>\"\n",
-    "\n",
-    "set_llm_cache(MongoDBAtlasSemanticCache(\n",
-    "    embedding=FakeEmbeddings(),\n",
-    "    connection_string=mongodb_atlas_uri,\n",
-    "    collection_name=COLLECTION_NAME,\n",
-    "    database_name=DATABASE_NAME,\n",
-    "))\n",
-    "```\n",
-    "\n",
-    "To find more resources about using MongoDBSemanticCache visit [here](https://www.mongodb.com/blog/post/introducing-semantic-caching-dedicated-mongodb-lang-chain-package-gen-ai-apps)"
-   ]
-  },
  {
   "cell_type": "markdown",
   "id": "726fe754",
@@ -1070,7 +993,7 @@
   "metadata": {},
   "outputs": [
    {
-     "name": "stdout",
+     "name": "stdin",
     "output_type": "stream",
     "text": [
      "CASSANDRA_KEYSPACE =  demo_keyspace\n"
@@ -1106,7 +1029,7 @@
   "metadata": {},
   "outputs": [
    {
-     "name": "stdout",
+     "name": "stdin",
     "output_type": "stream",
     "text": [
      "ASTRA_DB_ID =  01234567-89ab-cdef-0123-456789abcdef\n",
@@ -1710,12 +1633,14 @@
   "metadata": {},
   "outputs": [],
   "source": [
+    "from elasticsearch import Elasticsearch\n",
    "from langchain.globals import set_llm_cache\n",
    "from langchain_elasticsearch import ElasticsearchCache\n",
    "\n",
+    "es_client = Elasticsearch(hosts=\"http://localhost:9200\")\n",
    "set_llm_cache(\n",
    "    ElasticsearchCache(\n",
-    "        es_url=\"http://localhost:9200\",\n",
+    "        es_connection=es_client,\n",
    "        index_name=\"llm-chat-cache\",\n",
    "        metadata={\"project\": \"my_chatgpt_project\"},\n",
    "    )\n",
@@ -1759,6 +1684,7 @@
    "import json\n",
    "from typing import Any, Dict, List\n",
    "\n",
+    "from elasticsearch import Elasticsearch\n",
    "from langchain.globals import set_llm_cache\n",
    "from langchain_core.caches import RETURN_VAL_TYPE\n",
    "from langchain_elasticsearch import ElasticsearchCache\n",
@@ -1789,10 +1715,9 @@
    "        ]\n",
    "\n",
    "\n",
+    "es_client = Elasticsearch(hosts=\"http://localhost:9200\")\n",
    "set_llm_cache(\n",
-    "    SearchableElasticsearchCache(\n",
-    "        es_url=\"http://localhost:9200\", index_name=\"llm-chat-cache\"\n",
-    "    )\n",
+    "    SearchableElasticsearchCache(es_connection=es_client, index_name=\"llm-chat-cache\")\n",
    ")"
   ]
  },
@@ -2146,71 +2071,6 @@
    "# so it uses the cached result!\n",
    "llm(\"Tell me one joke\")"
   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "ae1f5e1c-085e-4998-9f2d-b5867d2c3d5b",
-   "metadata": {
-    "execution": {
-     "iopub.execute_input": "2024-05-31T17:18:43.345495Z",
-     "iopub.status.busy": "2024-05-31T17:18:43.345015Z",
-     "iopub.status.idle": "2024-05-31T17:18:43.351003Z",
-     "shell.execute_reply": "2024-05-31T17:18:43.350073Z",
-     "shell.execute_reply.started": "2024-05-31T17:18:43.345456Z"
-    }
-   },
-   "source": [
-    "## Cache classes: summary table"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "65072e45-10bc-40f1-979b-2617656bbbce",
-   "metadata": {
-    "execution": {
-     "iopub.execute_input": "2024-05-31T17:16:05.616430Z",
-     "iopub.status.busy": "2024-05-31T17:16:05.616221Z",
-     "iopub.status.idle": "2024-05-31T17:16:05.624164Z",
-     "shell.execute_reply": "2024-05-31T17:16:05.623673Z",
-     "shell.execute_reply.started": "2024-05-31T17:16:05.616418Z"
-    }
-   },
-   "source": [
-    "**Cache** classes are implemented by inheriting the [BaseCache](https://api.python.langchain.com/en/latest/caches/langchain_core.caches.BaseCache.html) class.\n",
-    "\n",
-    "This table lists all 20 derived classes with links to the API Reference.\n",
-    "\n",
-    "\n",
-    "| Namespace 🔻 | Class |\n",
-    "|------------|---------|\n",
-    "| langchain_astradb.cache | [AstraDBCache](https://api.python.langchain.com/en/latest/cache/langchain_astradb.cache.AstraDBCache.html) |\n",
-    "| langchain_astradb.cache | [AstraDBSemanticCache](https://api.python.langchain.com/en/latest/cache/langchain_astradb.cache.AstraDBSemanticCache.html) |\n",
-    "| langchain_community.cache | [AstraDBCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.AstraDBCache.html) |\n",
-    "| langchain_community.cache | [AstraDBSemanticCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.AstraDBSemanticCache.html) |\n",
-    "| langchain_community.cache | [AzureCosmosDBSemanticCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.AzureCosmosDBSemanticCache.html) |\n",
-    "| langchain_community.cache | [CassandraCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.CassandraCache.html) |\n",
-    "| langchain_community.cache | [CassandraSemanticCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.CassandraSemanticCache.html) |\n",
-    "| langchain_community.cache | [GPTCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.GPTCache.html) |\n",
-    "| langchain_community.cache | [InMemoryCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.InMemoryCache.html) |\n",
-    "| langchain_community.cache | [MomentoCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.MomentoCache.html) |\n",
-    "| langchain_community.cache | [OpenSearchSemanticCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.OpenSearchSemanticCache.html) |\n",
-    "| langchain_community.cache | [RedisSemanticCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.RedisSemanticCache.html) |\n",
-    "| langchain_community.cache | [SQLAlchemyCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.SQLAlchemyCache.html) |\n",
-    "| langchain_community.cache | [SQLAlchemyMd5Cache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.SQLAlchemyMd5Cache.html) |\n",
-    "| langchain_community.cache | [UpstashRedisCache](https://api.python.langchain.com/en/latest/cache/langchain_community.cache.UpstashRedisCache.html) |\n",
-    "| langchain_core.caches | [InMemoryCache](https://api.python.langchain.com/en/latest/caches/langchain_core.caches.InMemoryCache.html) |\n",
-    "| langchain_elasticsearch.cache | [ElasticsearchCache](https://api.python.langchain.com/en/latest/cache/langchain_elasticsearch.cache.ElasticsearchCache.html) |\n",
-    "| langchain_mongodb.cache | [MongoDBAtlasSemanticCache](https://api.python.langchain.com/en/latest/cache/langchain_mongodb.cache.MongoDBAtlasSemanticCache.html) |\n",
-    "| langchain_mongodb.cache | [MongoDBCache](https://api.python.langchain.com/en/latest/cache/langchain_mongodb.cache.MongoDBCache.html) |\n"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "19067f14-c69a-4156-9504-af43a0713669",
-   "metadata": {},
-   "outputs": [],
-   "source": []
  }
 ],
 "metadata": {
@@ -2229,7 +2089,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
-   "version": "3.10.12"
+   "version": "3.9.1"
  }
 },
 "nbformat": 4,
--- a/docs/docs/integrations/llms/aleph_alpha.ipynb
+++ b/docs/docs/integrations/llms/aleph_alpha.ipynb
@@ -12,17 +12,6 @@
    "This example goes over how to use LangChain to interact with Aleph Alpha models"
   ]
  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "84483bd5",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "# Installing the langchain package needed to use the integration\n",
-    "%pip install -qU langchain-community"
-   ]
-  },
  {
   "cell_type": "code",
   "execution_count": null,
--- a/docs/docs/integrations/llms/alibabacloud_pai_eas_endpoint.ipynb
+++ b/docs/docs/integrations/llms/alibabacloud_pai_eas_endpoint.ipynb
@@ -9,16 +9,6 @@
    ">[Machine Learning Platform for AI of Alibaba Cloud](https://www.alibabacloud.com/help/en/pai) is a machine learning or deep learning engineering platform intended for enterprises and developers. It provides easy-to-use, cost-effective, high-performance, and easy-to-scale plug-ins that can be applied to various industry scenarios. With over 140 built-in optimization algorithms, `Machine Learning Platform for AI` provides whole-process AI engineering capabilities including data labeling (`PAI-iTAG`), model building (`PAI-Designer` and `PAI-DSW`), model training (`PAI-DLC`), compilation optimization, and inference deployment (`PAI-EAS`). `PAI-EAS` supports different types of hardware resources, including CPUs and GPUs, and features high throughput and low latency. It allows you to deploy large-scale complex models with a few clicks and perform elastic scale-ins and scale-outs in real time. It also provides a comprehensive O&M and monitoring system."
   ]
  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "##Installing the langchain packages needed to use the integration\n",
-    "%pip install -qU langchain-community"
-   ]
-  },
  {
   "cell_type": "code",
   "execution_count": 8,
--- a/docs/docs/integrations/llms/amazon_api_gateway.ipynb
+++ b/docs/docs/integrations/llms/amazon_api_gateway.ipynb
@@ -16,16 +16,6 @@
    ">`API Gateway` handles all the tasks involved in accepting and processing up to hundreds of thousands of concurrent API calls, including traffic management, CORS support, authorization >and access control, throttling, monitoring, and API version management. `API Gateway` has no minimum fees or startup costs. You pay for the API calls you receive and the amount of data >transferred out and, with the `API Gateway` tiered pricing model, you can reduce your cost as your API usage scales."
   ]
  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "##Installing the langchain packages needed to use the integration\n",
-    "%pip install -qU langchain-community"
-   ]
-  },
  {
   "cell_type": "markdown",
   "metadata": {},
--- a/docs/docs/integrations/llms/anthropic.ipynb
+++ b/docs/docs/integrations/llms/anthropic.ipynb
@@ -3,15 +3,10 @@
  {
   "cell_type": "raw",
   "id": "602a52a4",
-   "metadata": {
-    "vscode": {
-     "languageId": "raw"
-    }
-   },
+   "metadata": {},
   "source": [
    "---\n",
    "sidebar_label: Anthropic\n",
-    "sidebar_class_name: hidden\n",
    "---"
   ]
  },
@@ -22,14 +17,10 @@
   "source": [
    "# AnthropicLLM\n",
    "\n",
-    ":::caution\n",
-    "You are currently on a page documenting the use of Anthropic legacy Claude 2 models as [text completion models](/docs/concepts/#llms). The latest and most popular Anthropic models are [chat completion models](/docs/concepts/#chat-models).\n",
-    "\n",
-    "You are probably looking for [this page instead](/docs/integrations/chat/anthropic/).\n",
-    ":::\n",
-    "\n",
    "This example goes over how to use LangChain to interact with `Anthropic` models.\n",
    "\n",
+    "NOTE: AnthropicLLM only supports legacy Claude 2 models. To use the newest Claude 3 models, please use [`ChatAnthropic`](/docs/integrations/chat/anthropic) instead.\n",
+    "\n",
    "## Installation"
   ]
  },
--- a/docs/docs/integrations/llms/anyscale.ipynb
+++ b/docs/docs/integrations/llms/anyscale.ipynb
@@ -12,17 +12,6 @@
    "This example goes over how to use LangChain to interact with [Anyscale Endpoint](https://app.endpoints.anyscale.com/). "
   ]
  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "134bd228",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "##Installing the langchain packages needed to use the integration\n",
-    "%pip install -qU langchain-community"
-   ]
-  },
  {
   "cell_type": "code",
   "execution_count": null,
--- a/docs/docs/integrations/llms/aphrodite.ipynb
+++ b/docs/docs/integrations/llms/aphrodite.ipynb
@@ -18,17 +18,6 @@
    "To use, you should have the `aphrodite-engine` python package installed."
   ]
  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "4dba1074",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "##Installing the langchain packages needed to use the integration\n",
-    "%pip install -qU langchain-community"
-   ]
-  },
  {
   "cell_type": "code",
   "execution_count": null,
--- a/docs/docs/integrations/llms/arcee.ipynb
+++ b/docs/docs/integrations/llms/arcee.ipynb
@@ -8,16 +8,6 @@
    "This notebook demonstrates how to use the `Arcee` class for generating text using Arcee's Domain Adapted Language Models (DALMs)."
   ]
  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "##Installing the langchain packages needed to use the integration\n",
-    "%pip install -qU langchain-community"
-   ]
-  },
  {
   "cell_type": "markdown",
   "metadata": {},
--- a/docs/docs/integrations/llms/azure_ml.ipynb
+++ b/docs/docs/integrations/llms/azure_ml.ipynb
@@ -11,16 +11,6 @@
    "This notebook goes over how to use an LLM hosted on an `Azure ML Online Endpoint`."
   ]
  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "##Installing the langchain packages needed to use the integration\n",
-    "%pip install -qU langchain-community"
-   ]
-  },
  {
   "cell_type": "code",
   "execution_count": null,
--- a/docs/docs/integrations/llms/azure_openai.ipynb
+++ b/docs/docs/integrations/llms/azure_openai.ipynb
@@ -7,13 +7,7 @@
   "source": [
    "# Azure OpenAI\n",
    "\n",
-    ":::caution\n",
-    "You are currently on a page documenting the use of Azure OpenAI [text completion models](/docs/concepts/#llms). The latest and most popular Azure OpenAI models are [chat completion models](/docs/concepts/#chat-models).\n",
-    "\n",
-    "Unless you are specifically using `gpt-3.5-turbo-instruct`, you are probably looking for [this page instead](/docs/integrations/chat/azure_chat_openai/).\n",
-    ":::\n",
-    "\n",
-    "This page goes over how to use LangChain with [Azure OpenAI](https://aka.ms/azure-openai).\n",
+    "This notebook goes over how to use Langchain with [Azure OpenAI](https://aka.ms/azure-openai).\n",
    "\n",
    "The Azure OpenAI API is compatible with OpenAI's API.  The `openai` Python package makes it easy to use both OpenAI and Azure OpenAI.  You can call Azure OpenAI the same way you call OpenAI with the exceptions noted below.\n",
    "\n",
--- a/docs/docs/integrations/llms/baichuan.ipynb
+++ b/docs/docs/integrations/llms/baichuan.ipynb
@@ -8,16 +8,6 @@
    "Baichuan Inc. (https://www.baichuan-ai.com/) is a Chinese startup in the era of AGI, dedicated to addressing fundamental human needs: Efficiency, Health, and Happiness."
   ]
  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "##Installing the langchain packages needed to use the integration\n",
-    "%pip install -qU langchain-community"
-   ]
-  },
  {
   "cell_type": "markdown",
   "metadata": {},
--- a/docs/docs/integrations/llms/baidu_qianfan_endpoint.ipynb
+++ b/docs/docs/integrations/llms/baidu_qianfan_endpoint.ipynb
@@ -45,16 +45,6 @@
    "- AquilaChat-7B"
   ]
  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "##Installing the langchain packages needed to use the integration\n",
-    "%pip install -qU langchain-community"
-   ]
-  },
  {
   "cell_type": "code",
   "execution_count": 2,
--- a/docs/docs/integrations/llms/banana.ipynb
+++ b/docs/docs/integrations/llms/banana.ipynb
@@ -12,16 +12,6 @@
    "This example goes over how to use LangChain to interact with Banana models"
   ]
  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "##Installing the langchain packages needed to use the integration\n",
-    "%pip install -qU  langchain-community"
-   ]
-  },
  {
   "cell_type": "code",
   "execution_count": null,
--- a/docs/docs/integrations/llms/baseten.ipynb
+++ b/docs/docs/integrations/llms/baseten.ipynb
@@ -45,16 +45,6 @@
    "In this example, we'll work with Mistral 7B. [Deploy Mistral 7B here](https://app.baseten.co/explore/mistral_7b_instruct) and follow along with the deployed model's ID, found in the model dashboard."
   ]
  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "##Installing the langchain packages needed to use the integration\n",
-    "%pip install -qU langchain-community"
-   ]
-  },
  {
   "cell_type": "code",
   "execution_count": null,
--- a/docs/docs/integrations/llms/bedrock.ipynb
+++ b/docs/docs/integrations/llms/bedrock.ipynb
@@ -11,12 +11,6 @@
   "cell_type": "markdown",
   "metadata": {},
   "source": [
-    ":::caution\n",
-    "You are currently on a page documenting the use of Amazon Bedrock models as [text completion models](/docs/concepts/#llms). Many popular models available on Bedrock are [chat completion models](/docs/concepts/#chat-models).\n",
-    "\n",
-    "You may be looking for [this page instead](/docs/integrations/chat/bedrock/).\n",
-    ":::\n",
-    "\n",
    ">[Amazon Bedrock](https://aws.amazon.com/bedrock/) is a fully managed service that offers a choice of \n",
    "> high-performing foundation models (FMs) from leading AI companies like `AI21 Labs`, `Anthropic`, `Cohere`, \n",
    "> `Meta`, `Stability AI`, and `Amazon` via a single API, along with a broad set of capabilities you need to \n",
--- a/docs/docs/integrations/llms/cohere.ipynb
+++ b/docs/docs/integrations/llms/cohere.ipynb
@@ -7,12 +7,6 @@
   "source": [
    "# Cohere\n",
    "\n",
-    ":::caution\n",
-    "You are currently on a page documenting the use of Cohere models as [text completion models](/docs/concepts/#llms). Many popular Cohere models are [chat completion models](/docs/concepts/#chat-models).\n",
-    "\n",
-    "You may be looking for [this page instead](/docs/integrations/chat/cohere/).\n",
-    ":::\n",
-    "\n",
    ">[Cohere](https://cohere.ai/about) is a Canadian startup that provides natural language processing models that help companies improve human-machine interactions.\n",
    "\n",
    "Head to the [API reference](https://api.python.langchain.com/en/latest/llms/langchain_community.llms.cohere.Cohere.html) for detailed documentation of all attributes and methods."
@@ -199,7 +193,7 @@
   "id": "39198f7d-6fc8-4662-954a-37ad38c4bec4",
   "metadata": {},
   "source": [
-    "You can also easily combine with a prompt template for easy structuring of user input. We can do this using [LCEL](/docs/concepts#langchain-expression-language-lcel)"
+    "You can also easily combine with a prompt template for easy structuring of user input. We can do this using [LCEL](/docs/concepts#langchain-expression-language)"
   ]
  },
  {
--- a/docs/docs/integrations/llms/fireworks.ipynb
+++ b/docs/docs/integrations/llms/fireworks.ipynb
@@ -7,12 +7,6 @@
   "source": [
    "# Fireworks\n",
    "\n",
-    ":::caution\n",
-    "You are currently on a page documenting the use of Fireworks models as [text completion models](/docs/concepts/#llms). Many popular Fireworks models are [chat completion models](/docs/concepts/#chat-models).\n",
-    "\n",
-    "You may be looking for [this page instead](/docs/integrations/chat/fireworks/).\n",
-    ":::\n",
-    "\n",
    ">[Fireworks](https://app.fireworks.ai/) accelerates product development on generative AI by creating an innovative AI experiment and production platform. \n",
    "\n",
    "This example goes over how to use LangChain to interact with `Fireworks` models."
--- a/docs/docs/integrations/llms/google_ai.ipynb
+++ b/docs/docs/integrations/llms/google_ai.ipynb
@@ -25,12 +25,6 @@
   "id": "bead5ede-d9cc-44b9-b062-99c90a10cf40",
   "metadata": {},
   "source": [
-    ":::caution\n",
-    "You are currently on a page documenting the use of Google models as [text completion models](/docs/concepts/#llms). Many popular Google models are [chat completion models](/docs/concepts/#chat-models).\n",
-    "\n",
-    "You may be looking for [this page instead](/docs/integrations/chat/google_generative_ai/).\n",
-    ":::\n",
-    "\n",
    "A guide on using [Google Generative AI](https://developers.generativeai.google/) models with Langchain. Note: It's separate from Google Cloud Vertex AI [integration](/docs/integrations/llms/google_vertex_ai_palm)."
   ]
  },
--- a/docs/docs/integrations/llms/google_vertex_ai_palm.ipynb
+++ b/docs/docs/integrations/llms/google_vertex_ai_palm.ipynb
@@ -15,12 +15,6 @@
   "source": [
    "# Google Cloud Vertex AI\n",
    "\n",
-    ":::caution\n",
-    "You are currently on a page documenting the use of Google Vertex [text completion models](/docs/concepts/#llms). Many Google models are [chat completion models](/docs/concepts/#chat-models).\n",
-    "\n",
-    "You may be looking for [this page instead](/docs/integrations/chat/google_vertex_ai_palm/).\n",
-    ":::\n",
-    "\n",
    "**Note:** This is separate from the `Google Generative AI` integration, it exposes [Vertex AI Generative API](https://cloud.google.com/vertex-ai/docs/generative-ai/learn/overview) on `Google Cloud`.\n",
    "\n",
    "VertexAI exposes all foundational models available in google cloud:\n",
@@ -334,7 +328,7 @@
   "cell_type": "markdown",
   "metadata": {},
   "source": [
-    "You can also easily combine with a prompt template for easy structuring of user input. We can do this using [LCEL](/docs/concepts#langchain-expression-language-lcel)"
+    "You can also easily combine with a prompt template for easy structuring of user input. We can do this using [LCEL](/docs/concepts#langchain-expression-language)"
   ]
  },
  {
--- a/docs/docs/integrations/llms/ollama.ipynb
+++ b/docs/docs/integrations/llms/ollama.ipynb
@@ -6,12 +6,6 @@
   "source": [
    "# Ollama\n",
    "\n",
-    ":::caution\n",
-    "You are currently on a page documenting the use of Ollama models as [text completion models](/docs/concepts/#llms). Many popular Ollama models are [chat completion models](/docs/concepts/#chat-models).\n",
-    "\n",
-    "You may be looking for [this page instead](/docs/integrations/chat/ollama/).\n",
-    ":::\n",
-    "\n",
    "[Ollama](https://ollama.ai/) allows you to run open-source large language models, such as Llama 2, locally.\n",
    "\n",
    "Ollama bundles model weights, configuration, and data into a single package, defined by a Modelfile. \n",
--- a/docs/docs/integrations/llms/openai.ipynb
+++ b/docs/docs/integrations/llms/openai.ipynb
@@ -7,12 +7,6 @@
   "source": [
    "# OpenAI\n",
    "\n",
-    ":::caution\n",
-    "You are currently on a page documenting the use of OpenAI [text completion models](/docs/concepts/#llms). The latest and most popular OpenAI models are [chat completion models](/docs/concepts/#chat-models).\n",
-    "\n",
-    "Unless you are specifically using `gpt-3.5-turbo-instruct`, you are probably looking for [this page instead](/docs/integrations/chat/openai/).\n",
-    ":::\n",
-    "\n",
    "[OpenAI](https://platform.openai.com/docs/introduction) offers a spectrum of models with different levels of power suitable for different tasks.\n",
    "\n",
    "This example goes over how to use LangChain to interact with `OpenAI` [models](https://platform.openai.com/docs/models)"
--- a/docs/docs/integrations/llms/together.ipynb
+++ b/docs/docs/integrations/llms/together.ipynb
@@ -7,12 +7,6 @@
   "source": [
    "# Together AI\n",
    "\n",
-    ":::caution\n",
-    "You are currently on a page documenting the use of Together AI models as [text completion models](/docs/concepts/#llms). Many popular Together AI models are [chat completion models](/docs/concepts/#chat-models).\n",
-    "\n",
-    "You may be looking for [this page instead](/docs/integrations/chat/together/).\n",
-    ":::\n",
-    "\n",
    "[Together AI](https://www.together.ai/) offers an API to query [50+ leading open-source models](https://docs.together.ai/docs/inference-models) in a couple lines of code.\n",
    "\n",
    "This example goes over how to use LangChain to interact with Together AI models."
--- a/docs/docs/integrations/memory/kafka_chat_message_history.ipynb
+++ b/docs/docs/integrations/memory/kafka_chat_message_history.ipynb
@@ -1,245 +0,0 @@
-{
- "cells": [
-  {
-   "cell_type": "markdown",
-   "id": "c21deb80-9cf7-4185-8205-a38110152d2c",
-   "metadata": {},
-   "source": [
-    "# Kafka\n",
-    "\n",
-    "[Kafka](https://github.com/apache/kafka) is a distributed messaging system that is used to publish and subscribe to streams of records. \n",
-    "This demo shows how to use `KafkaChatMessageHistory` to store and retrieve chat messages from a Kafka cluster."
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "c7c4fc02-18ac-4285-b8d6-507357e2aa13",
-   "metadata": {},
-   "source": [
-    "A running Kafka cluster is required to run the demo. You can follow this [instruction](https://developer.confluent.io/get-started/python) to create a Kafka cluster locally."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "id": "f09f3b45-c4ff-4e59-bf79-238cc85d6465",
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "from langchain_community.chat_message_histories import KafkaChatMessageHistory\n",
-    "\n",
-    "chat_session_id = \"chat-message-history-kafka\"\n",
-    "bootstrap_servers = \"localhost:64797\"  # host:port. `localhost:Plaintext Ports` if setup Kafka cluster locally\n",
-    "history = KafkaChatMessageHistory(\n",
-    "    chat_session_id,\n",
-    "    bootstrap_servers,\n",
-    ")"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "109812d2-85c5-4a65-a8a0-2d16eb80347b",
-   "metadata": {},
-   "source": [
-    "Optional parameters to construct `KafkaChatMessageHistory`:\n",
-    " - `ttl_ms`: Time to live in milliseconds for the chat messages.\n",
-    " - `partition`: Number of partition of the topic to store the chat messages.\n",
-    " - `replication_factor`: Replication factor of the topic to store the chat messages."
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "c8fba39f-650b-4192-94ea-1a2a89f5348d",
-   "metadata": {},
-   "source": [
-    "`KafkaChatMessageHistory` internally uses Kafka consumer to read chat messages, and it has the ability to mark the consumed position persistently. It has following methods to retrieve chat messages:\n",
-    "- `messages`: continue consuming chat messages from last one.\n",
-    "- `messages_from_beginning`: reset the consumer to the beginning of the history and consume messages. Optional parameters:\n",
-    "    1. `max_message_count`: maximum number of messages to read.\n",
-    "    2. `max_time_sec`: maximum time in seconds to read messages.\n",
-    "- `messages_from_latest`: reset the consumer to the end of the chat history and try consuming messages. Optional parameters same as above.\n",
-    "- `messages_from_last_consumed`: return messages continuing from the last consumed message, similar to `messages`, but with optional parameters.\n",
-    "\n",
-    "`max_message_count` and `max_time_sec` are used to avoid blocking indefinitely when retrieving messages.\n",
-    "As a result, `messages` and other methods to retrieve messages may not return all messages in the chat history. You will need to specify `max_message_count` and `max_time_sec` to retrieve all chat history in a single batch.\n"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "caf2176b-db7a-451a-a292-3d9fde585ded",
-   "metadata": {},
-   "source": [
-    "Add messages and retrieve."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 3,
-   "id": "7e52a70d-3921-4614-b8cd-53b8d3c2deb4",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[HumanMessage(content='hi!'), AIMessage(content='whats up?')]"
-      ]
-     },
-     "execution_count": 3,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "history.add_user_message(\"hi!\")\n",
-    "history.add_ai_message(\"whats up?\")\n",
-    "\n",
-    "history.messages"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "874ce388-da8f-4796-b9ca-3ac114195b10",
-   "metadata": {},
-   "source": [
-    "Calling `messages` again returns an empty list because the consumer is at the end of the chat history."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 4,
-   "id": "f863618e-7da1-4f46-9182-7a1387b93b16",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[]"
-      ]
-     },
-     "execution_count": 4,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "history.messages"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "e108255b-c240-44f7-9ecc-52bf04cd15b6",
-   "metadata": {},
-   "source": [
-    "Add new messages and continue consuming."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 5,
-   "id": "31aa7403-5392-4ad4-ba43-226020a274e3",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[HumanMessage(content='hi again!'), AIMessage(content='whats up again?')]"
-      ]
-     },
-     "execution_count": 5,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "history.add_user_message(\"hi again!\")\n",
-    "history.add_ai_message(\"whats up again?\")\n",
-    "history.messages"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "5062fabc-c605-40dd-933b-c68de2727874",
-   "metadata": {},
-   "source": [
-    "To reset the consumer and read from beginning:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 6,
-   "id": "005816ae-c8ed-4e41-9ecd-b6432578c8f1",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[HumanMessage(content='hi again!'),\n",
-       " AIMessage(content='whats up again?'),\n",
-       " HumanMessage(content='hi!'),\n",
-       " AIMessage(content='whats up?')]"
-      ]
-     },
-     "execution_count": 6,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "history.messages_from_beginning()"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "id": "42cc7bed-5cd7-417f-94fd-fe2e511cc9c6",
-   "metadata": {},
-   "source": [
-    "Set the consumer to the end of the chat history, add a couple of new messages, and consume:"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 7,
-   "id": "d8b4f1cd-fa47-461b-b1b6-278ad54e9ac5",
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "[HumanMessage(content='HI!'), AIMessage(content='WHATS UP?')]"
-      ]
-     },
-     "execution_count": 7,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "history.messages_from_latest()\n",
-    "history.add_user_message(\"HI!\")\n",
-    "history.add_ai_message(\"WHATS UP?\")\n",
-    "history.messages"
-   ]
-  }
- ],
- "metadata": {
-  "kernelspec": {
-   "display_name": "Python 3 (ipykernel)",
-   "language": "python",
-   "name": "python3"
-  },
-  "language_info": {
-   "codemirror_mode": {
-    "name": "ipython",
-    "version": 3
-   },
-   "file_extension": ".py",
-   "mimetype": "text/x-python",
-   "name": "python",
-   "nbconvert_exporter": "python",
-   "pygments_lexer": "ipython3",
-   "version": "3.8.18"
-  }
- },
- "nbformat": 4,
- "nbformat_minor": 5
-}
--- a/docs/docs/integrations/platforms/anthropic.mdx
+++ b/docs/docs/integrations/platforms/anthropic.mdx
@@ -14,18 +14,6 @@ pip install -U langchain-anthropic
 You need to set the `ANTHROPIC_API_KEY` environment variable.
 You can get an Anthropic API key [here](https://console.anthropic.com/settings/keys)

-## Chat Models
-
-### ChatAnthropic
-
-See a [usage example](/docs/integrations/chat/anthropic).
-
-```python
-from langchain_anthropic import ChatAnthropic
-
-model = ChatAnthropic(model='claude-3-opus-20240229')
-```
-
 ## LLMs

 ### [Legacy] AnthropicLLM
@@ -40,3 +28,17 @@ from langchain_anthropic import AnthropicLLM

 model = AnthropicLLM(model='claude-2.1')
 ```
+
+## Chat Models
+
+### ChatAnthropic
+
+See a [usage example](/docs/integrations/chat/anthropic).
+
+```python
+from langchain_anthropic import ChatAnthropic
+
+model = ChatAnthropic(model='claude-3-opus-20240229')
+```
+
+
--- a/docs/docs/integrations/platforms/aws.mdx
+++ b/docs/docs/integrations/platforms/aws.mdx
@@ -134,20 +134,6 @@ See a [usage example](/docs/integrations/document_loaders/athena).
 from langchain_community.document_loaders.athena import AthenaLoader
 ```

-### AWS Glue
-
->The [AWS Glue Data Catalog](https://docs.aws.amazon.com/en_en/glue/latest/dg/catalog-and-crawler.html) is a centralized metadata 
-> repository that allows you to manage, access, and share metadata about 
-> your data stored in AWS. It acts as a metadata store for your data assets, 
-> enabling various AWS services and your applications to query and connect 
-> to the data they need efficiently.
-
-See a [usage example](/docs/integrations/document_loaders/glue_catalog).
-
-```python
-from langchain_community.document_loaders.glue_catalog import GlueCatalogLoader
-```
-
 ## Vector stores

 ### Amazon OpenSearch Service
--- a/docs/docs/integrations/platforms/google.mdx
+++ b/docs/docs/integrations/platforms/google.mdx
@@ -2,10 +2,45 @@

 All functionality related to [Google Cloud Platform](https://cloud.google.com/) and other `Google` products.

-## Chat models
+## LLMs

 We recommend individual developers to start with Gemini API (`langchain-google-genai`) and move to Vertex AI (`langchain-google-vertexai`) when they need access to commercial support and higher rate limits. If you’re already Cloud-friendly or Cloud-native, then you can get started in Vertex AI straight away.
-Please see [here](https://ai.google.dev/gemini-api/docs/migrate-to-cloud) for more information.
+Please, find more information [here](https://ai.google.dev/gemini-api/docs/migrate-to-cloud).
+
+### Google Generative AI
+
+Access GoogleAI `Gemini` models such as `gemini-pro` and `gemini-pro-vision` through the `GoogleGenerativeAI` class.
+
+Install python package.
+
+```bash
+pip install langchain-google-genai
+```
+
+See a [usage example](/docs/integrations/llms/google_ai).
+
+```python
+from langchain_google_genai import GoogleGenerativeAI
+```
+
+### Vertex AI Model Garden
+
+Access `PaLM` and hundreds of OSS models via `Vertex AI Model Garden` service.
+
+We need to install `langchain-google-vertexai` python package.
+
+```bash
+pip install langchain-google-vertexai
+```
+
+See a [usage example](/docs/integrations/llms/google_vertex_ai_palm#vertex-model-garden).
+
+```python
+from langchain_google_vertexai import VertexAIModelGarden
+```
+
+
+## Chat models

 ### Google Generative AI

@@ -72,40 +107,6 @@ See a [usage example](/docs/integrations/chat/google_vertex_ai_palm).
 from langchain_google_vertexai import ChatVertexAI
 ```

-## LLMs
-
-### Google Generative AI
-
-Access GoogleAI `Gemini` models such as `gemini-pro` and `gemini-pro-vision` through the `GoogleGenerativeAI` class.
-
-Install python package.
-
-```bash
-pip install langchain-google-genai
-```
-
-See a [usage example](/docs/integrations/llms/google_ai).
-
-```python
-from langchain_google_genai import GoogleGenerativeAI
-```
-
-### Vertex AI Model Garden
-
-Access `PaLM` and hundreds of OSS models via `Vertex AI Model Garden` service.
-
-We need to install `langchain-google-vertexai` python package.
-
-```bash
-pip install langchain-google-vertexai
-```
-
-See a [usage example](/docs/integrations/llms/google_vertex_ai_palm#vertex-model-garden).
-
-```python
-from langchain_google_vertexai import VertexAIModelGarden
-```
-
 ## Embedding models

 ### Google Generative AI Embeddings
--- a/docs/docs/integrations/platforms/index.mdx
+++ b/docs/docs/integrations/platforms/index.mdx
@@ -24,7 +24,6 @@ These providers have standalone `langchain-{provider}` packages for improved ver
 - [Anthropic](/docs/integrations/platforms/anthropic)
 - [Astra DB](/docs/integrations/providers/astradb)
 - [Cohere](/docs/integrations/providers/cohere)
- [Couchbase](/docs/integrations/providers/couchbase)
 - [Elasticsearch](/docs/integrations/providers/elasticsearch)
 - [Exa Search](/docs/integrations/providers/exa_search)
 - [Fireworks](/docs/integrations/providers/fireworks)
--- a/docs/docs/integrations/platforms/microsoft.mdx
+++ b/docs/docs/integrations/platforms/microsoft.mdx
@@ -6,6 +6,24 @@ keywords: [azure]

 All functionality related to `Microsoft Azure` and other `Microsoft` products.

+## LLMs
+
+### Azure ML
+
+See a [usage example](/docs/integrations/llms/azure_ml).
+
+```python
+from langchain_community.llms.azureml_endpoint import AzureMLOnlineEndpoint
+```
+
+### Azure OpenAI
+
+See a [usage example](/docs/integrations/llms/azure_openai).
+
+```python
+from langchain_openai import AzureOpenAI
+```
+
 ## Chat Models
 ### Azure OpenAI

@@ -33,24 +51,6 @@ See a [usage example](/docs/integrations/chat/azure_chat_openai)
 from langchain_openai import AzureChatOpenAI
 ```

-## LLMs
-
-### Azure ML
-
-See a [usage example](/docs/integrations/llms/azure_ml).
-
-```python
-from langchain_community.llms.azureml_endpoint import AzureMLOnlineEndpoint
-```
-
-### Azure OpenAI
-
-See a [usage example](/docs/integrations/llms/azure_openai).
-
-```python
-from langchain_openai import AzureOpenAI
-```
-
 ## Embedding Models
 ### Azure OpenAI

@@ -225,7 +225,7 @@ from langchain_community.document_loaders.onenote import OneNoteLoader

 ## Vector stores

-### Azure Cosmos DB MongoDB vCore
+### Azure Cosmos DB

 >[Azure Cosmos DB for MongoDB vCore](https://learn.microsoft.com/en-us/azure/cosmos-db/mongodb/vcore/) makes it easy to create a database with full native MongoDB support.
 > You can apply your MongoDB experience and continue to use your favorite MongoDB drivers, SDKs, and tools by pointing your application to the API for MongoDB vCore account's connection string.
@@ -255,38 +255,6 @@ See a [usage example](/docs/integrations/vectorstores/azure_cosmos_db).
 from langchain_community.vectorstores import AzureCosmosDBVectorSearch
 ```

-### Azure Cosmos DB NoSQL
-
->[Azure Cosmos DB for NoSQL](https://learn.microsoft.com/en-us/azure/cosmos-db/nosql/vector-search) now offers vector indexing and search in preview.
-This feature is designed to handle high-dimensional vectors, enabling efficient and accurate vector search at any scale. You can now store vectors
-directly in the documents alongside your data. This means that each document in your database can contain not only traditional schema-free data,
-but also high-dimensional vectors as other properties of the documents. This colocation of data and vectors allows for efficient indexing and searching,
-as the vectors are stored in the same logical unit as the data they represent. This simplifies data management, AI application architectures, and the
-efficiency of vector-based operations.
-
-#### Installation and Setup
-
-See [detail configuration instructions](/docs/integrations/vectorstores/azure_cosmos_db_no_sql).
-
-We need to install `azure-cosmos` python package.
-
-```bash
-pip install azure-cosmos
-```
-
-#### Deploy Azure Cosmos DB on Microsoft Azure
-
-Azure Cosmos DB offers a solution for modern apps and intelligent workloads by being very responsive with dynamic and elastic autoscale. It is available
-in every Azure region and can automatically replicate data closer to users. It has SLA guaranteed low-latency and high availability.
-
-[Sign Up](https://learn.microsoft.com/en-us/azure/cosmos-db/nosql/quickstart-python?pivots=devcontainer-codespace) for free to get started today.
-
-See a [usage example](/docs/integrations/vectorstores/azure_cosmos_db_no_sql).
-
-```python
-from langchain_community.vectorstores import AzureCosmosDBNoSQLVectorSearch
-```
-
 ## Retrievers
 ### Azure AI Search

--- a/docs/docs/integrations/platforms/openai.mdx
+++ b/docs/docs/integrations/platforms/openai.mdx
@@ -25,19 +25,6 @@ pip install langchain-openai

 Get an OpenAI api key and set it as an environment variable (`OPENAI_API_KEY`)

-## Chat model
-
-See a [usage example](/docs/integrations/chat/openai).
-
-```python
-from langchain_openai import ChatOpenAI
-```
-
-If you are using a model hosted on `Azure`, you should use different wrapper for that:
-```python
-from langchain_openai import AzureChatOpenAI
-```
-For a more detailed walkthrough of the `Azure` wrapper, see [here](/docs/integrations/chat/azure_chat_openai).

 ## LLM

@@ -51,7 +38,21 @@ If you are using a model hosted on `Azure`, you should use different wrapper for
 ```python
 from langchain_openai import AzureOpenAI
 ```
-For a more detailed walkthrough of the `Azure` wrapper, see [here](/docs/integrations/llms/azure_openai).
+For a more detailed walkthrough of the `Azure` wrapper, see [here](/docs/integrations/llms/azure_openai)
+
+## Chat model
+
+See a [usage example](/docs/integrations/chat/openai).
+
+```python
+from langchain_openai import ChatOpenAI
+```
+
+If you are using a model hosted on `Azure`, you should use different wrapper for that:
+```python
+from langchain_openai import AzureChatOpenAI
+```
+For a more detailed walkthrough of the `Azure` wrapper, see [here](/docs/integrations/chat/azure_chat_openai)

 ## Embedding Model

--- a/docs/docs/integrations/providers/cohere.mdx
+++ b/docs/docs/integrations/providers/cohere.mdx
@@ -48,9 +48,6 @@ print(llm.invoke("Come up with a pet name"))
 Usage of the Cohere (legacy) [LLM model](/docs/integrations/llms/cohere) 
 ### ReAct Agent

-The agent is based on the paper
-[ReAct: Synergizing Reasoning and Acting in Language Models](https://arxiv.org/abs/2210.03629).
-
 ```python
 from langchain_community.tools.tavily_search import TavilySearchResults
 from langchain_cohere import ChatCohere, create_cohere_react_agent
--- a/docs/docs/integrations/providers/couchbase.mdx
+++ b/docs/docs/integrations/providers/couchbase.mdx
@@ -6,19 +6,12 @@

 ## Installation and Setup

-We have to install the `langchain-couchbase` package.
+We have to install the `couchbase`package.

 ```bash
-pip install langchain-couchbase
+pip install couchbase
 ```

-## Vector Store
-
-See a [usage example](/docs/integrations/vectorstores/couchbase).
-
-```python
-from langchain_couchbase import CouchbaseVectorStore
-```

 ## Document loader

--- a/docs/docs/integrations/providers/elasticsearch.mdx
+++ b/docs/docs/integrations/providers/elasticsearch.mdx
@@ -32,7 +32,7 @@ pip install langchain-elasticsearch
 See a [usage example](/docs/integrations/text_embedding/elasticsearch).

 ```python
-from langchain_elasticsearch import ElasticsearchEmbeddings
+from langchain_elasticsearch.embeddings import ElasticsearchEmbeddings
 ```

 ## Vector store
@@ -40,7 +40,7 @@ from langchain_elasticsearch import ElasticsearchEmbeddings
 See a [usage example](/docs/integrations/vectorstores/elasticsearch).

 ```python
-from langchain_elasticsearch import ElasticsearchStore
+from langchain_elasticsearch.vectorstores import ElasticsearchStore
 ```


@@ -49,21 +49,5 @@ from langchain_elasticsearch import ElasticsearchStore
 See a [usage example](/docs/integrations/memory/elasticsearch_chat_message_history).

 ```python
-from langchain_elasticsearch import ElasticsearchChatMessageHistory
-```
-
-## LLM cache
-
-See a [usage example](/docs/integrations/llm_caching/#elasticsearch-cache).
-
-```python
-from langchain_elasticsearch import ElasticsearchCache
-```
-
-## Byte Store
-
-See a [usage example](/docs/integrations/stores/elasticsearch).
-
-```python
-from langchain_elasticsearch import ElasticsearchEmbeddingsCache
+from langchain_elasticsearch.chat_history import ElasticsearchChatMessageHistory
 ```
--- a/docs/docs/integrations/providers/nvidia.mdx
+++ b/docs/docs/integrations/providers/nvidia.mdx
@@ -62,7 +62,7 @@ When ready to deploy, you can self-host models with NVIDIA NIM—which is includ
 from langchain_nvidia_ai_endpoints import ChatNVIDIA, NVIDIAEmbeddings, NVIDIARerank

 # connect to an chat NIM running at localhost:8000, specifyig a specific model
-llm = ChatNVIDIA(base_url="http://localhost:8000/v1", model="meta/llama3-8b-instruct")
+llm = ChatNVIDIA(base_url="http://localhost:8000/v1", model="meta-llama3-8b-instruct")

 # connect to an embedding NIM running at localhost:8080
 embedder = NVIDIAEmbeddings(base_url="http://localhost:8080/v1")
--- a/docs/docs/integrations/providers/premai.md
+++ b/docs/docs/integrations/providers/premai.md
@@ -113,21 +113,29 @@ The diameters of individual galaxies range from 80,000-150,000 light-years.
 {
    "document_chunks": [
        {
-            "repository_id": 19xx,
-            "document_id": 13xx,
-            "chunk_id": 173xxx,
+            "repository_id": 1991,
+            "document_id": 1307,
+            "chunk_id": 173926,
            "document_name": "Kegy 202 Chapter 2",
            "similarity_score": 0.586126983165741,
            "content": "n thousands\n                                                                                                                                               of           light-years. The diameters of individual\n                                                                                                                                               galaxies range from 80,000-150,000 light\n                                                                                                                       "
        },
        {
-            "repository_id": 19xx,
-            "document_id": 13xx,
-            "chunk_id": 173xxx,
+            "repository_id": 1991,
+            "document_id": 1307,
+            "chunk_id": 173925,
            "document_name": "Kegy 202 Chapter 2",
            "similarity_score": 0.4815782308578491,
            "content": "                                                for development of galaxies. A galaxy contains\n                                                                                                                                               a large number of stars. Galaxies spread over\n                                                                                                                                               vast distances that are measured in thousands\n                                       "
        },
+        {
+            "repository_id": 1991,
+            "document_id": 1307,
+            "chunk_id": 173916,
+            "document_name": "Kegy 202 Chapter 2",
+            "similarity_score": 0.38112708926200867,
+            "content": " was separated from the               from each other as the balloon expands.\n  solar surface. As the passing star moved away,             Similarly, the distance between the galaxies is\n  the material separated from the solar surface\n  continued to revolve around the sun and it\n  slowly condensed into planets. Sir James Jeans\n  and later Sir Harold Jeffrey supported thisnot to be republishedalso found to be increasing and thereby, the\n                                                             universe is"
+        }
    ]
 }
 ```
@@ -165,42 +173,7 @@ This will stream tokens one after the other.

 > Please note: As of now, RAG with streaming is not supported. However we still support it with our API. You can learn more about that [here](https://docs.premai.io/get-started/chat-completion-sse). 

-
-## Prem Templates
-
-Writing Prompt Templates can be super messy. Prompt templates are long, hard to manage, and must be continuously tweaked to improve and keep the same throughout the application. 
-
-With **Prem**, writing and managing prompts can be super easy. The **_Templates_** tab inside the [launchpad](https://docs.premai.io/get-started/launchpad) helps you write as many prompts you need and use it inside the SDK to make your application running using those prompts. You can read more about Prompt Templates [here](https://docs.premai.io/get-started/prem-templates). 
-
-To use Prem Templates natively with LangChain, you need to pass an id the `HumanMessage`. This id should be the name the variable of your prompt template. the `content` in `HumanMessage` should be the value of that variable. 
-
-let's say for example, if your prompt template was this:
-
-```text
-Say hello to my name and say a feel-good quote
-from my age. My name is: {name} and age is {age}
-```
-
-So now your human_messages should look like:
-
-```python
-human_messages = [
-    HumanMessage(content="Shawn", id="name"),
-    HumanMessage(content="22", id="age")
-]
-```
-
-Pass this `human_messages` to ChatPremAI Client. Please note: Do not forget to
-pass the additional `template_id` to invoke generation with Prem Templates. If you are not aware of `template_id` you can learn more about that [in our docs](https://docs.premai.io/get-started/prem-templates). Here is an example:
-
-```python
-template_id = "78069ce8-xxxxx-xxxxx-xxxx-xxx"
-response = chat.invoke([human_message], template_id=template_id)
-```
-
-Prem Templates are also available for Streaming too. 
-
-## Prem Embeddings
+## PremEmbeddings

 In this section we are going to dicuss how we can get access to different embedding model using `PremEmbeddings` with LangChain. Lets start by importing our modules and setting our API Key. 

--- a/docs/docs/integrations/providers/vearch.md
+++ b/docs/docs/integrations/providers/vearch.md
@@ -8,7 +8,7 @@ Vearch Python SDK enables vearch to use locally. Vearch python sdk can be instal

 # Vectorstore

-Vearch also can used as vectorstore. Most details in [this notebook](/docs/integrations/vectorstores/vearch)
+Vearch also can used as vectorstore. Most detalis in [this notebook](/docs/integrations/vectorstores/vearch)

 ```python
 from langchain_community.vectorstores import Vearch
--- a/docs/docs/integrations/retrievers/kinetica.ipynb
+++ b/docs/docs/integrations/retrievers/kinetica.ipynb
@@ -22,7 +22,7 @@
   "outputs": [],
   "source": [
    "# Please ensure that this connector is installed in your working environment.\n",
-    "%pip install gpudb==7.2.0.9"
+    "%pip install gpudb==7.2.0.1"
   ]
  },
  {
--- a/docs/docs/integrations/text_embedding/index.mdx
+++ b/docs/docs/integrations/text_embedding/index.mdx
@@ -1,114 +0,0 @@
---
-sidebar_position: 0
-sidebar_class_name: hidden
---
-
-# Embedding models
-
-**Embedding model** classes are implemented by inheriting the [Embeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_core.embeddings.Embeddings.html) class.
-
-This table lists all 100 derived classes.
-
-
-| Namespace 🔻 | Class |
-|------------|---------|
-| langchain.chains.hyde.base | [HypotheticalDocumentEmbedder](https://api.python.langchain.com/en/latest/chains/langchain.chains.hyde.base.HypotheticalDocumentEmbedder.html) |
-| langchain.embeddings.cache | [CacheBackedEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain.embeddings.cache.CacheBackedEmbeddings.html) |
-| langchain_ai21.embeddings | [AI21Embeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_ai21.embeddings.AI21Embeddings.html) |
-| langchain_aws.embeddings.bedrock | [BedrockEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_aws.embeddings.bedrock.BedrockEmbeddings.html) |
-| langchain_cohere.embeddings | [CohereEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_cohere.embeddings.CohereEmbeddings.html) |
-| langchain_community.embeddings.aleph_alpha | [AlephAlphaAsymmetricSemanticEmbedding](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.aleph_alpha.AlephAlphaAsymmetricSemanticEmbedding.html) |
-| langchain_community.embeddings.aleph_alpha | [AlephAlphaSymmetricSemanticEmbedding](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.aleph_alpha.AlephAlphaSymmetricSemanticEmbedding.html) |
-| langchain_community.embeddings.anyscale | [AnyscaleEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.anyscale.AnyscaleEmbeddings.html) |
-| langchain_community.embeddings.awa | [AwaEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.awa.AwaEmbeddings.html) |
-| langchain_community.embeddings.azure_openai | [AzureOpenAIEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.azure_openai.AzureOpenAIEmbeddings.html) |
-| langchain_community.embeddings.baichuan | [BaichuanTextEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.baichuan.BaichuanTextEmbeddings.html) |
-| langchain_community.embeddings.baidu_qianfan_endpoint | [QianfanEmbeddingsEndpoint](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.baidu_qianfan_endpoint.QianfanEmbeddingsEndpoint.html) |
-| langchain_community.embeddings.bedrock | [BedrockEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.bedrock.BedrockEmbeddings.html) |
-| langchain_community.embeddings.bookend | [BookendEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.bookend.BookendEmbeddings.html) |
-| langchain_community.embeddings.clarifai | [ClarifaiEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.clarifai.ClarifaiEmbeddings.html) |
-| langchain_community.embeddings.cloudflare_workersai | [CloudflareWorkersAIEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.cloudflare_workersai.CloudflareWorkersAIEmbeddings.html) |
-| langchain_community.embeddings.clova | [ClovaEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.clova.ClovaEmbeddings.html) |
-| langchain_community.embeddings.cohere | [CohereEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.cohere.CohereEmbeddings.html) |
-| langchain_community.embeddings.dashscope | [DashScopeEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.dashscope.DashScopeEmbeddings.html) |
-| langchain_community.embeddings.databricks | [DatabricksEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.databricks.DatabricksEmbeddings.html) |
-| langchain_community.embeddings.deepinfra | [DeepInfraEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.deepinfra.DeepInfraEmbeddings.html) |
-| langchain_community.embeddings.edenai | [EdenAiEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.edenai.EdenAiEmbeddings.html) |
-| langchain_community.embeddings.elasticsearch | [ElasticsearchEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.elasticsearch.ElasticsearchEmbeddings.html) |
-| langchain_community.embeddings.embaas | [EmbaasEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.embaas.EmbaasEmbeddings.html) |
-| langchain_community.embeddings.ernie | [ErnieEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.ernie.ErnieEmbeddings.html) |
-| langchain_community.embeddings.fake | [DeterministicFakeEmbedding](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.fake.DeterministicFakeEmbedding.html) |
-| langchain_community.embeddings.fake | [FakeEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.fake.FakeEmbeddings.html) |
-| langchain_community.embeddings.fastembed | [FastEmbedEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.fastembed.FastEmbedEmbeddings.html) |
-| langchain_community.embeddings.gigachat | [GigaChatEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.gigachat.GigaChatEmbeddings.html) |
-| langchain_community.embeddings.google_palm | [GooglePalmEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.google_palm.GooglePalmEmbeddings.html) |
-| langchain_community.embeddings.gpt4all | [GPT4AllEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.gpt4all.GPT4AllEmbeddings.html) |
-| langchain_community.embeddings.gradient_ai | [GradientEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.gradient_ai.GradientEmbeddings.html) |
-| langchain_community.embeddings.huggingface | [HuggingFaceBgeEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.huggingface.HuggingFaceBgeEmbeddings.html) |
-| langchain_community.embeddings.huggingface | [HuggingFaceEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.huggingface.HuggingFaceEmbeddings.html) |
-| langchain_community.embeddings.huggingface | [HuggingFaceInferenceAPIEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.huggingface.HuggingFaceInferenceAPIEmbeddings.html) |
-| langchain_community.embeddings.huggingface | [HuggingFaceInstructEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.huggingface.HuggingFaceInstructEmbeddings.html) |
-| langchain_community.embeddings.huggingface_hub | [HuggingFaceHubEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.huggingface_hub.HuggingFaceHubEmbeddings.html) |
-| langchain_community.embeddings.infinity | [InfinityEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.infinity.InfinityEmbeddings.html) |
-| langchain_community.embeddings.infinity_local | [InfinityEmbeddingsLocal](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.infinity_local.InfinityEmbeddingsLocal.html) |
-| langchain_community.embeddings.ipex_llm | [IpexLLMBgeEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.ipex_llm.IpexLLMBgeEmbeddings.html) |
-| langchain_community.embeddings.itrex | [QuantizedBgeEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.itrex.QuantizedBgeEmbeddings.html) |
-| langchain_community.embeddings.javelin_ai_gateway | [JavelinAIGatewayEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.javelin_ai_gateway.JavelinAIGatewayEmbeddings.html) |
-| langchain_community.embeddings.jina | [JinaEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.jina.JinaEmbeddings.html) |
-| langchain_community.embeddings.johnsnowlabs | [JohnSnowLabsEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.johnsnowlabs.JohnSnowLabsEmbeddings.html) |
-| langchain_community.embeddings.laser | [LaserEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.laser.LaserEmbeddings.html) |
-| langchain_community.embeddings.llamacpp | [LlamaCppEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.llamacpp.LlamaCppEmbeddings.html) |
-| langchain_community.embeddings.llamafile | [LlamafileEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.llamafile.LlamafileEmbeddings.html) |
-| langchain_community.embeddings.llm_rails | [LLMRailsEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.llm_rails.LLMRailsEmbeddings.html) |
-| langchain_community.embeddings.localai | [LocalAIEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.localai.LocalAIEmbeddings.html) |
-| langchain_community.embeddings.minimax | [MiniMaxEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.minimax.MiniMaxEmbeddings.html) |
-| langchain_community.embeddings.mlflow | [MlflowCohereEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.mlflow.MlflowCohereEmbeddings.html) |
-| langchain_community.embeddings.mlflow | [MlflowEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.mlflow.MlflowEmbeddings.html) |
-| langchain_community.embeddings.mlflow_gateway | [MlflowAIGatewayEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.mlflow_gateway.MlflowAIGatewayEmbeddings.html) |
-| langchain_community.embeddings.modelscope_hub | [ModelScopeEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.modelscope_hub.ModelScopeEmbeddings.html) |
-| langchain_community.embeddings.mosaicml | [MosaicMLInstructorEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.mosaicml.MosaicMLInstructorEmbeddings.html) |
-| langchain_community.embeddings.nemo | [NeMoEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.nemo.NeMoEmbeddings.html) |
-| langchain_community.embeddings.nlpcloud | [NLPCloudEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.nlpcloud.NLPCloudEmbeddings.html) |
-| langchain_community.embeddings.oci_generative_ai | [OCIGenAIEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.oci_generative_ai.OCIGenAIEmbeddings.html) |
-| langchain_community.embeddings.octoai_embeddings | [OctoAIEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.octoai_embeddings.OctoAIEmbeddings.html) |
-| langchain_community.embeddings.ollama | [OllamaEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.ollama.OllamaEmbeddings.html) |
-| langchain_community.embeddings.openai | [OpenAIEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.openai.OpenAIEmbeddings.html) |
-| langchain_community.embeddings.openvino | [OpenVINOBgeEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.openvino.OpenVINOBgeEmbeddings.html) |
-| langchain_community.embeddings.openvino | [OpenVINOEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.openvino.OpenVINOEmbeddings.html) |
-| langchain_community.embeddings.optimum_intel | [QuantizedBiEncoderEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.optimum_intel.QuantizedBiEncoderEmbeddings.html) |
-| langchain_community.embeddings.oracleai | [OracleEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.oracleai.OracleEmbeddings.html) |
-| langchain_community.embeddings.ovhcloud | [OVHCloudEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.ovhcloud.OVHCloudEmbeddings.html) |
-| langchain_community.embeddings.premai | [PremAIEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.premai.PremAIEmbeddings.html) |
-| langchain_community.embeddings.sagemaker_endpoint | [SagemakerEndpointEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.sagemaker_endpoint.SagemakerEndpointEmbeddings.html) |
-| langchain_community.embeddings.sambanova | [SambaStudioEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.sambanova.SambaStudioEmbeddings.html) |
-| langchain_community.embeddings.self_hosted | [SelfHostedEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.self_hosted.SelfHostedEmbeddings.html) |
-| langchain_community.embeddings.self_hosted_hugging_face | [SelfHostedHuggingFaceEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.self_hosted_hugging_face.SelfHostedHuggingFaceEmbeddings.html) |
-| langchain_community.embeddings.self_hosted_hugging_face | [SelfHostedHuggingFaceInstructEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.self_hosted_hugging_face.SelfHostedHuggingFaceInstructEmbeddings.html) |
-| langchain_community.embeddings.solar | [SolarEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.solar.SolarEmbeddings.html) |
-| langchain_community.embeddings.spacy_embeddings | [SpacyEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.spacy_embeddings.SpacyEmbeddings.html) |
-| langchain_community.embeddings.sparkllm | [SparkLLMTextEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.sparkllm.SparkLLMTextEmbeddings.html) |
-| langchain_community.embeddings.tensorflow_hub | [TensorflowHubEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.tensorflow_hub.TensorflowHubEmbeddings.html) |
-| langchain_community.embeddings.text2vec | [Text2vecEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.text2vec.Text2vecEmbeddings.html) |
-| langchain_community.embeddings.titan_takeoff | [TitanTakeoffEmbed](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.titan_takeoff.TitanTakeoffEmbed.html) |
-| langchain_community.embeddings.vertexai | [VertexAIEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.vertexai.VertexAIEmbeddings.html) |
-| langchain_community.embeddings.volcengine | [VolcanoEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.volcengine.VolcanoEmbeddings.html) |
-| langchain_community.embeddings.voyageai | [VoyageEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.voyageai.VoyageEmbeddings.html) |
-| langchain_community.embeddings.xinference | [XinferenceEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.xinference.XinferenceEmbeddings.html) |
-| langchain_community.embeddings.yandex | [YandexGPTEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.yandex.YandexGPTEmbeddings.html) |
-| langchain_community.embeddings.zhipuai | [ZhipuAIEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_community.embeddings.zhipuai.ZhipuAIEmbeddings.html) |
-| langchain_core.embeddings.fake | [DeterministicFakeEmbedding](https://api.python.langchain.com/en/latest/embeddings/langchain_core.embeddings.fake.DeterministicFakeEmbedding.html) |
-| langchain_core.embeddings.fake | [FakeEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_core.embeddings.fake.FakeEmbeddings.html) |
-| langchain_elasticsearch.embeddings | [ElasticsearchEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_elasticsearch.embeddings.ElasticsearchEmbeddings.html) |
-| langchain_fireworks.embeddings | [FireworksEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_fireworks.embeddings.FireworksEmbeddings.html) |
-| langchain_google_genai.embeddings | [GoogleGenerativeAIEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_google_genai.embeddings.GoogleGenerativeAIEmbeddings.html) |
-| langchain_google_genai.google_vector_store | [ServerSideEmbedding](https://api.python.langchain.com/en/latest/google_vector_store/langchain_google_genai.google_vector_store.ServerSideEmbedding.html) |
-| langchain_google_vertexai.embeddings | [VertexAIEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_google_vertexai.embeddings.VertexAIEmbeddings.html) |
-| langchain_huggingface.embeddings.huggingface | [HuggingFaceEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_huggingface.embeddings.huggingface.HuggingFaceEmbeddings.html) |
-| langchain_huggingface.embeddings.huggingface_endpoint | [HuggingFaceEndpointEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_huggingface.embeddings.huggingface_endpoint.HuggingFaceEndpointEmbeddings.html) |
-| langchain_ibm.embeddings | [WatsonxEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_ibm.embeddings.WatsonxEmbeddings.html) |
-| langchain_mistralai.embeddings | [MistralAIEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_mistralai.embeddings.MistralAIEmbeddings.html) |
-| langchain_nomic.embeddings | [NomicEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_nomic.embeddings.NomicEmbeddings.html) |
-| langchain_nvidia_ai_endpoints.embeddings | [NVIDIAEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_nvidia_ai_endpoints.embeddings.NVIDIAEmbeddings.html) |
-| langchain_together.embeddings | [TogetherEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_together.embeddings.TogetherEmbeddings.html) |
-| langchain_upstage.embeddings | [UpstageEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_upstage.embeddings.UpstageEmbeddings.html) |
-| langchain_voyageai.embeddings | [VoyageAIEmbeddings](https://api.python.langchain.com/en/latest/embeddings/langchain_voyageai.embeddings.VoyageAIEmbeddings.html) |
--- a/docs/docs/integrations/text_embedding/ovhcloud.ipynb
+++ b/docs/docs/integrations/text_embedding/ovhcloud.ipynb
--- a/docs/docs/integrations/text_embedding/voyageai.ipynb
+++ b/docs/docs/integrations/text_embedding/voyageai.ipynb
@@ -33,9 +33,7 @@
    "- `voyage-code-2`\n",
    "- `voyage-2`\n",
    "- `voyage-law-2`\n",
-    "- `voyage-large-2-instruct`\n",
-    "- `voyage-finance-2`\n",
-    "- `voyage-multilingual-2`"
+    "- `voyage-large-2-instruct`"
   ]
  },
  {
--- a/docs/docs/integrations/tools/apify.ipynb
+++ b/docs/docs/integrations/tools/apify.ipynb
@@ -42,16 +42,14 @@
   "source": [
    "from langchain.indexes import VectorstoreIndexCreator\n",
    "from langchain_community.utilities import ApifyWrapper\n",
-    "from langchain_core.documents import Document\n",
-    "from langchain_openai import OpenAI\n",
-    "from langchain_openai.embeddings import OpenAIEmbeddings"
+    "from langchain_core.documents import Document"
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
-    "Initialize it using your [Apify API token](https://docs.apify.com/platform/integrations/api#api-token) and for the purpose of this example, also with your OpenAI API key:"
+    "Initialize it using your [Apify API token](https://console.apify.com/account/integrations) and for the purpose of this example, also with your OpenAI API key:"
   ]
  },
  {
@@ -105,7 +103,7 @@
   "metadata": {},
   "outputs": [],
   "source": [
-    "index = VectorstoreIndexCreator(embedding=OpenAIEmbeddings()).from_loaders([loader])"
+    "index = VectorstoreIndexCreator().from_loaders([loader])"
   ]
  },
  {
@@ -122,7 +120,7 @@
   "outputs": [],
   "source": [
    "query = \"What is LangChain?\"\n",
-    "result = index.query_with_sources(query, llm=OpenAI())"
+    "result = index.query_with_sources(query)"
   ]
  },
  {
@@ -162,7 +160,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
-   "version": "3.11.3"
+   "version": "3.10.1"
  }
 },
 "nbformat": 4,
--- a/docs/docs/integrations/tools/bing_search.ipynb
+++ b/docs/docs/integrations/tools/bing_search.ipynb
@@ -6,17 +6,16 @@
   "source": [
    "# Bing Search\n",
    "\n",
-    "[Microsoft Bing](https://www.bing.com/), commonly referred to as `Bing` or `Bing Search`, is a web search engine owned and operated by `Microsoft`."
+    ">[Microsoft Bing](https://www.bing.com/), commonly referred to as `Bing` or `Bing Search`, is a web search engine owned and operated by `Microsoft`."
   ]
  },
  {
   "cell_type": "markdown",
   "metadata": {},
   "source": [
-    "## Setup\n",
-    "Following the [instruction](https://learn.microsoft.com/en-us/bing/search-apis/bing-web-search/create-bing-search-service-resource) to create Azure Bing Search v7 service, and get the subscription key\n",
+    "This notebook goes over how to use the bing search component.\n",
    "\n",
-    "The integration lives in the `langchain-community` package."
+    "Then we will need to set some environment variables."
   ]
  },
  {
@@ -25,25 +24,24 @@
   "metadata": {},
   "outputs": [],
   "source": [
-    "%pip install -U langchain-community"
+    "%pip install --upgrade --quiet langchain-community"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 2,
+   "execution_count": 20,
   "metadata": {},
   "outputs": [],
   "source": [
-    "import getpass\n",
    "import os\n",
    "\n",
-    "os.environ[\"BING_SUBSCRIPTION_KEY\"] = getpass.getpass()\n",
+    "os.environ[\"BING_SUBSCRIPTION_KEY\"] = \"<key>\"\n",
    "os.environ[\"BING_SEARCH_URL\"] = \"https://api.bing.microsoft.com/v7.0/search\""
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 3,
+   "execution_count": 21,
   "metadata": {},
   "outputs": [],
   "source": [
@@ -52,25 +50,25 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 4,
+   "execution_count": 22,
   "metadata": {},
   "outputs": [],
   "source": [
-    "search = BingSearchAPIWrapper(k=4)"
+    "search = BingSearchAPIWrapper()"
   ]
  },
  {
   "cell_type": "code",
-   "execution_count": 18,
+   "execution_count": 23,
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
-       "'<b>Python is a</b> versatile and powerful language that lets you work quickly and integrate systems more effectively. Learn how to get started, download the latest version, access documentation, find jobs, and join the Python community. <b>Python is a</b> popular programming language for various purposes. Find the latest version of Python for different operating systems, download release notes, and learn about the development process. Learn <b>Python,</b> a popular programming language for web applications, with examples, exercises, and references. Get certified by completing the PYTHON <b>course</b> at W3Schools. Learn the basic concepts and features of <b>Python,</b> a powerful and easy to learn programming language. The tutorial covers topics such as data structures, modules, classes, exceptions, input and output, and more. Learn why and how to use <b>Python,</b> a popular and easy-to-learn programming language. Find installation guides, tutorials, documentation, resources and FAQs for beginners and experienced programmers. Learn about <b>Python,</b> a high-level, general-purpose programming language with a focus on code readability and multiple paradigms. Find out its history, design, features, libraries, implementations, popularity, uses, and influences. Real <b>Python</b> offers tutorials, books, courses, and news for <b>Python</b> developers of all skill levels. Whether you want to learn <b>Python</b> basics, web development, data science, or machine learning, you can find useful articles and code examples here. Learn how to install, use, and extend <b>Python</b> 3.12.3, a popular programming language. Find tutorials, library references, API guides, FAQs, and more. <b>Python</b> is a powerful, fast, friendly and open-source language that runs everywhere. Learn how to get started, explore applications, join the community and access the latest news and events. Learn the basics of <b>Python</b> programming language with examples of numbers, text, variables, and operators. This tutorial covers the syntax, types, and features of <b>Python</b> for beginners.'"
+       "'Thanks to the flexibility of <b>Python</b> and the powerful ecosystem of packages, the Azure CLI supports features such as autocompletion (in shells that support it), persistent credentials, JMESPath result parsing, lazy initialization, network-less unit tests, and more. Building an open-source and cross-platform Azure CLI with <b>Python</b> by Dan Taylor. <b>Python</b> releases by version number: Release version Release date Click for more. <b>Python</b> 3.11.1 Dec. 6, 2022 Download Release Notes. <b>Python</b> 3.10.9 Dec. 6, 2022 Download Release Notes. <b>Python</b> 3.9.16 Dec. 6, 2022 Download Release Notes. <b>Python</b> 3.8.16 Dec. 6, 2022 Download Release Notes. <b>Python</b> 3.7.16 Dec. 6, 2022 Download Release Notes. In this lesson, we will look at the += operator in <b>Python</b> and see how it works with several simple examples.. The operator ‘+=’ is a shorthand for the addition assignment operator.It adds two values and assigns the sum to a variable (left operand). W3Schools offers free online tutorials, references and exercises in all the major languages of the web. Covering popular subjects like HTML, CSS, JavaScript, <b>Python</b>, SQL, Java, and many, many more. This tutorial introduces the reader informally to the basic concepts and features of the <b>Python</b> language and system. It helps to have a <b>Python</b> interpreter handy for hands-on experience, but all examples are self-contained, so the tutorial can be read off-line as well. For a description of standard objects and modules, see The <b>Python</b> Standard ... <b>Python</b> is a general-purpose, versatile, and powerful programming language. It&#39;s a great first language because <b>Python</b> code is concise and easy to read. Whatever you want to do, <b>python</b> can do it. From web development to machine learning to data science, <b>Python</b> is the language for you. To install <b>Python</b> using the Microsoft Store: Go to your Start menu (lower left Windows icon), type &quot;Microsoft Store&quot;, select the link to open the store. Once the store is open, select Search from the upper-right menu and enter &quot;<b>Python</b>&quot;. Select which version of <b>Python</b> you would like to use from the results under Apps. Under the “<b>Python</b> Releases for Mac OS X” heading, click the link for the Latest <b>Python</b> 3 Release - <b>Python</b> 3.x.x. As of this writing, the latest version was <b>Python</b> 3.8.4. Scroll to the bottom and click macOS 64-bit installer to start the download. When the installer is finished downloading, move on to the next step. Step 2: Run the Installer'"
      ]
     },
-     "execution_count": 18,
+     "execution_count": 23,
     "metadata": {},
     "output_type": "execute_result"
    }
@@ -89,7 +87,7 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 6,
+   "execution_count": 24,
   "metadata": {},
   "outputs": [],
   "source": [
@@ -98,16 +96,16 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 7,
+   "execution_count": 25,
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
-       "'<b>Python</b> is a versatile and powerful language that lets you work quickly and integrate systems more effectively. Learn how to get started, download the latest version, access documentation, find jobs, and join the Python community.'"
+       "'Thanks to the flexibility of <b>Python</b> and the powerful ecosystem of packages, the Azure CLI supports features such as autocompletion (in shells that support it), persistent credentials, JMESPath result parsing, lazy initialization, network-less unit tests, and more. Building an open-source and cross-platform Azure CLI with <b>Python</b> by Dan Taylor.'"
      ]
     },
-     "execution_count": 7,
+     "execution_count": 25,
     "metadata": {},
     "output_type": "execute_result"
    }
@@ -136,7 +134,7 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 8,
+   "execution_count": 26,
   "metadata": {},
   "outputs": [],
   "source": [
@@ -145,30 +143,27 @@
  },
  {
   "cell_type": "code",
-   "execution_count": 9,
+   "execution_count": 27,
   "metadata": {},
   "outputs": [
    {
     "data": {
      "text/plain": [
-       "[{'snippet': 'Learn about the nutrients, antioxidants, and potential health effects of<b> apples.</b> Find out how<b> apples</b> may help with weight loss, diabetes, heart disease, and cancer.',\n",
+       "[{'snippet': 'Lady Alice. Pink Lady <b>apples</b> aren’t the only lady in the apple family. Lady Alice <b>apples</b> were discovered growing, thanks to bees pollinating, in Washington. They are smaller and slightly more stout in appearance than other varieties. Their skin color appears to have red and yellow stripes running from stem to butt.',\n",
+       "  'title': '25 Types of Apples - Jessica Gavin',\n",
+       "  'link': 'https://www.jessicagavin.com/types-of-apples/'},\n",
+       " {'snippet': '<b>Apples</b> can do a lot for you, thanks to plant chemicals called flavonoids. And they have pectin, a fiber that breaks down in your gut. If you take off the apple’s skin before eating it, you won ...',\n",
+       "  'title': 'Apples: Nutrition &amp; Health Benefits - WebMD',\n",
+       "  'link': 'https://www.webmd.com/food-recipes/benefits-apples'},\n",
+       " {'snippet': '<b>Apples</b> boast many vitamins and minerals, though not in high amounts. However, <b>apples</b> are usually a good source of vitamin C. Vitamin C. Also called ascorbic acid, this vitamin is a common ...',\n",
       "  'title': 'Apples 101: Nutrition Facts and Health Benefits',\n",
       "  'link': 'https://www.healthline.com/nutrition/foods/apples'},\n",
-       " {'snippet': 'Learn how<b> apples</b> can improve your health with their fiber, antioxidants, and phytochemicals. Find out the best types of<b> apples</b> for different purposes, how to buy and store them, and what side effects to watch out for.',\n",
-       "  'title': 'Apples: Nutrition and Health Benefits - WebMD',\n",
-       "  'link': 'https://www.webmd.com/food-recipes/benefits-apples'},\n",
-       " {'snippet': '<b>Apples</b> are nutritious, filling, and versatile fruits that may lower your risk of various diseases. Learn how<b> apples</b> can support your weight loss, heart health, gut health, and brain health with scientific evidence.',\n",
-       "  'title': '10 Impressive Health Benefits of Apples',\n",
-       "  'link': 'https://www.healthline.com/nutrition/10-health-benefits-of-apples'},\n",
-       " {'snippet': 'An apple is a round, edible fruit produced by an apple tree (Malus spp., among them the domestic or orchard apple; Malus domestica).Apple trees are cultivated worldwide and are the most widely grown species in the genus Malus.The tree originated in Central Asia, where its wild ancestor, Malus sieversii, is still found.<b>Apples</b> have been grown for thousands of years in Eurasia and were introduced ...',\n",
-       "  'title': 'Apple - Wikipedia',\n",
-       "  'link': 'https://en.wikipedia.org/wiki/Apple'},\n",
-       " {'snippet': 'Learn about the most popular and diverse<b> apples</b> in the world, from ambrosia to winesap, with photos and descriptions. Find out their origins, flavors, uses, and nutritional benefits in this comprehensive guide to<b> apples.</b>',\n",
-       "  'title': '29 Types Of Apples From A to Z (With Photos!) - Live Eat Learn',\n",
-       "  'link': 'https://www.liveeatlearn.com/types-of-apples/'}]"
+       " {'snippet': 'Weight management. The fibers in <b>apples</b> can slow digestion, helping one to feel greater satisfaction after eating. After following three large prospective cohorts of 133,468 men and women for 24 years, researchers found that higher intakes of fiber-rich fruits with a low glycemic load, particularly <b>apples</b> and pears, were associated with the least amount of weight gain over time.',\n",
+       "  'title': 'Apples | The Nutrition Source | Harvard T.H. Chan School of Public Health',\n",
+       "  'link': 'https://www.hsph.harvard.edu/nutritionsource/food-features/apples/'}]"
      ]
     },
-     "execution_count": 9,
+     "execution_count": 27,
     "metadata": {},
     "output_type": "execute_result"
    }
@@ -176,148 +171,6 @@
   "source": [
    "search.results(\"apples\", 5)"
   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "## Tool Usage"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 10,
-   "metadata": {},
-   "outputs": [
-    {
-     "data": {
-      "text/plain": [
-       "BingSearchResults(api_wrapper=BingSearchAPIWrapper(bing_subscription_key='<your subscription key>', bing_search_url='https://api.bing.microsoft.com/v7.0/search', k=10, search_kwargs={}))"
-      ]
-     },
-     "execution_count": 10,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "import os\n",
-    "\n",
-    "from langchain_community.tools.bing_search import BingSearchResults\n",
-    "from langchain_community.utilities import BingSearchAPIWrapper\n",
-    "\n",
-    "api_wrapper = BingSearchAPIWrapper()\n",
-    "tool = BingSearchResults(api_wrapper=api_wrapper)\n",
-    "tool"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 12,
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "{'snippet': 'This chart shows the 14 day <b>weather</b> trend for 33.74°N 89.59°E with daily <b>weather</b> symbols, minimum and maximum temperatures, precipitation amount and probability.. The deviance is coloured within the temperature graph. The stronger the ups and downs, the more uncertain the forecast will be.', 'title': '14 Day Weather 33.74°N 89.59°E - meteoblue', 'link': 'https://www.meteoblue.com/en/weather/14-days/33.739N89.594E6216_Asia%2FShanghai'}\n",
-      "{'snippet': 'Get the monthly <b>weather</b> forecast for Huangpu District, <b>Shanghai</b>, China, including daily high/low, historical averages, to help you plan ahead.', 'title': 'Huangpu District, Shanghai, China Monthly Weather | AccuWeather', 'link': 'https://www.accuweather.com/en/cn/huangpu-district/60782/june-weather/60782'}\n",
-      "{'snippet': '<b>Shanghai</b> Hongqiao Airport is 60 miles from 31°31&#39;42.7&quot;N, 120°24&#39;12.7&quot;E, so the actual climate in 31°31&#39;42.7&quot;N, 120°24&#39;12.7&quot;E can vary a bit. Based on <b>weather</b> reports collected during 1992–2021. Showing: All Year January February March April May June July August September October November December', 'title': 'Climate &amp; Weather Averages in 31°31&#39;42.7&quot;N, 120°24&#39;12.7&quot;E, China', 'link': 'https://www.timeanddate.com/weather/@31.52853,120.40355/climate'}\n",
-      "{'snippet': 'Air Quality gives information using <b>weather</b> conditions, pollutants, and research from The <b>Weather</b> Channel and <b>weather</b>.com ... Today&#39;s Air Quality-<b>Shanghai</b>, People&#39;s Republic of China. 76.', 'title': 'Shanghai, People&#39;s Republic of China Weather', 'link': 'https://weather.com/forecast/air-quality/l/80415bb74d7ded500f89b569c51da53325719ddea6e1394485ad846e812e61d2'}\n"
-     ]
-    }
-   ],
-   "source": [
-    "# .invoke wraps utility.results\n",
-    "response = tool.invoke(\"What is the weather in Shanghai\")\n",
-    "for item in list(response):\n",
-    "    print(item)"
-   ]
-  },
-  {
-   "cell_type": "markdown",
-   "metadata": {},
-   "source": [
-    "## Chaining\n",
-    "\n",
-    "We show here how to use it as part of an [agent](/docs/tutorials/agents). We use the OpenAI Functions Agent, so we will need to setup and install the required dependencies for that. We will also use [LangSmith Hub](https://smith.langchain.com/hub) to pull the prompt from, so we will need to install that."
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": null,
-   "metadata": {},
-   "outputs": [],
-   "source": [
-    "# you need a model to use in the chain\n",
-    "%pip install --upgrade --quiet langchain langchain-openai langchainhub langchain-community"
-   ]
-  },
-  {
-   "cell_type": "code",
-   "execution_count": 15,
-   "metadata": {},
-   "outputs": [
-    {
-     "name": "stdout",
-     "output_type": "stream",
-     "text": [
-      "\n",
-      "\n",
-      "\u001b[1m> Entering new AgentExecutor chain...\u001b[0m\n",
-      "\u001b[32;1m\u001b[1;3m\n",
-      "Invoking: `bing_search_results_json` with `{'query': 'latest burning man floods'}`\n",
-      "\n",
-      "\n",
-      "\u001b[0m\u001b[36;1m\u001b[1;3m[{'snippet': 'Some festivalgoers have shared stories of their successful 6-mile hikes away from <b>Burning</b> <b>Man</b>. The worst of the rain Sunday is expected between 12 p.m. and 4 p.m. local time (3 p.m. to 7 p.m. ET ...', 'title': 'Live updates: Burning Man festival rain strands thousands in ... - CNN', 'link': 'https://www.cnn.com/us/live-news/nevada-desert-burning-man-weather-rain-09-03-23/index.html'}, {'snippet': 'Black Rock Forest, where around 70,000 <b>Burning</b> <b>Man</b> attendees are gathered for the festival, is in the northwest. &quot;Flash <b>flooding</b> caused by excessive rainfall&quot; is possible in parts of eastern ...', 'title': 'Burning Man flooding keeps thousands stranded at Nevada site as ...', 'link': 'https://www.nbcnews.com/news/us-news/live-blog/live-updates-burning-man-flooding-keeps-thousands-stranded-nevada-site-rcna103193'}, {'snippet': 'Thousands of <b>Burning</b> <b>Man</b> attendees finally made their mass exodus after intense rain over the weekend flooded camp sites and filled them with thick, ankle-deep mud – stranding more than 70,000 ...', 'title': 'Burning Man attendees make a mass exodus after a dramatic weekend ... - CNN', 'link': 'https://www.cnn.com/2023/09/05/us/burning-man-storms-shelter-exodus-tuesday/index.html'}, {'snippet': 'You’re going to get stuck,” hosts on <b>Burning</b> <b>Man</b> Information Radio, broadcasting from within the event, told festivalgoers early on Sept. 4. According to NBC News, the Pershing County Sheriff ...', 'title': 'Burning Man attendees make mass exodus after being stranded in ... - TODAY', 'link': 'https://www.today.com/news/what-is-burning-man-flood-death-rcna103231'}]\u001b[0m\u001b[32;1m\u001b[1;3mThe latest Burning Man festival experienced heavy rain and flooding, which resulted in thousands of festivalgoers being stranded. Some attendees had to hike for miles to safety. The rain caused flash flooding in parts of the festival site, including the Black Rock Forest where the event takes place. Campsites were flooded, and thick mud made movement difficult. Eventually, after the intense rain over the weekend, attendees were able to make a mass exodus from the festival. The Pershing County Sheriff's Office warned festivalgoers about the flooding and encouraged them to leave for safety.\u001b[0m\n",
-      "\n",
-      "\u001b[1m> Finished chain.\u001b[0m\n"
-     ]
-    },
-    {
-     "data": {
-      "text/plain": [
-       "{'input': 'What happened in the latest burning man floods?',\n",
-       " 'output': \"The latest Burning Man festival experienced heavy rain and flooding, which resulted in thousands of festivalgoers being stranded. Some attendees had to hike for miles to safety. The rain caused flash flooding in parts of the festival site, including the Black Rock Forest where the event takes place. Campsites were flooded, and thick mud made movement difficult. Eventually, after the intense rain over the weekend, attendees were able to make a mass exodus from the festival. The Pershing County Sheriff's Office warned festivalgoers about the flooding and encouraged them to leave for safety.\"}"
-      ]
-     },
-     "execution_count": 15,
-     "metadata": {},
-     "output_type": "execute_result"
-    }
-   ],
-   "source": [
-    "import getpass\n",
-    "import os\n",
-    "\n",
-    "from langchain import hub\n",
-    "from langchain.agents import AgentExecutor, create_openai_functions_agent\n",
-    "from langchain_openai import AzureChatOpenAI\n",
-    "\n",
-    "os.environ[\"AZURE_OPENAI_API_KEY\"] = getpass.getpass()\n",
-    "os.environ[\"AZURE_OPENAI_ENDPOINT\"] = \"https://<your-endpoint>.openai.azure.com/\"\n",
-    "os.environ[\"AZURE_OPENAI_API_VERSION\"] = \"2023-06-01-preview\"\n",
-    "os.environ[\"AZURE_OPENAI_DEPLOYMENT_NAME\"] = \"<your-deployment-name>\"\n",
-    "\n",
-    "instructions = \"\"\"You are an assistant.\"\"\"\n",
-    "base_prompt = hub.pull(\"langchain-ai/openai-functions-template\")\n",
-    "prompt = base_prompt.partial(instructions=instructions)\n",
-    "llm = AzureChatOpenAI(\n",
-    "    openai_api_key=os.environ[\"AZURE_OPENAI_API_KEY\"],\n",
-    "    azure_endpoint=os.environ[\"AZURE_OPENAI_ENDPOINT\"],\n",
-    "    azure_deployment=os.environ[\"AZURE_OPENAI_DEPLOYMENT_NAME\"],\n",
-    "    openai_api_version=os.environ[\"AZURE_OPENAI_API_VERSION\"],\n",
-    ")\n",
-    "tool = BingSearchResults(api_wrapper=api_wrapper)\n",
-    "tools = [tool]\n",
-    "agent = create_openai_functions_agent(llm, tools, prompt)\n",
-    "agent_executor = AgentExecutor(\n",
-    "    agent=agent,\n",
-    "    tools=tools,\n",
-    "    verbose=True,\n",
-    ")\n",
-    "agent_executor.invoke({\"input\": \"What happened in the latest burning man floods?\"})"
-   ]
  }
 ],
 "metadata": {
@@ -336,7 +189,7 @@
   "name": "python",
   "nbconvert_exporter": "python",
   "pygments_lexer": "ipython3",
-   "version": "3.10.14"
+   "version": "3.10.12"
  },
  "vscode": {
   "interpreter": {
--- a/Show More
+++ b/Show More
Author	SHA1	Message	Date
Lance Martin	3c6cb273d0	Add agent	2024-06-12 21:53:05 -07:00
Lance Martin	ad624ae3ed	Confirm tool use	2024-06-11 11:27:17 -07:00
Lance Martin	550cf6ac8f	Use OAI fxns for message processing	2024-06-10 07:37:03 -07:00
Lance Martin	5ab3d5e54c	Test structured output	2024-06-07 15:43:17 -07:00
Lance Martin	d4b2876e50	Fix code, test	2024-06-07 15:22:06 -07:00
Lance Martin	4cfadef235	Testing llamacpp	2024-06-07 12:21:36 -07:00