From cd80a805b2afa8b68564e71ee0ff00d5dc0a7cd6 Mon Sep 17 00:00:00 2001 From: Deepak Bhagat <149673145+deepak0x@users.noreply.github.com> Date: Tue, 28 Apr 2026 00:38:12 +0530 Subject: [PATCH] fix(text-splitters): remove invalid and duplicate separators in Kotlin, Rust, and Haskell (#37039) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Summary Fixes four issues in `get_separators_for_language()` in `character.py`: - **Kotlin**: removed `"\ncase "` — `case` is not a Kotlin keyword. Kotlin uses `when` expressions (already present in the list). This was copied from Java/Swift. - **Rust**: removed duplicate `"\nconst "` — appeared twice, once under function definitions and again under control flow statements. - **Haskell**: removed duplicate `"\n:: "` — appeared under function definitions and again under type declarations. - **Haskell**: removed duplicate `"\ndata "` — appeared under type declarations and again under record field declarations. All four are dead separators that never match or produce redundant splits. ## Issue Closes #37038 ## Types of changes - [x] Bug fix ## Checklist - [x] I have read the CONTRIBUTING doc - [x] Lint and unit tests pass locally with my changes --- libs/text-splitters/langchain_text_splitters/character.py | 4 ---- 1 file changed, 4 deletions(-) diff --git a/libs/text-splitters/langchain_text_splitters/character.py b/libs/text-splitters/langchain_text_splitters/character.py index dd2378a787..0e2e1ab2a9 100644 --- a/libs/text-splitters/langchain_text_splitters/character.py +++ b/libs/text-splitters/langchain_text_splitters/character.py @@ -266,7 +266,6 @@ class RecursiveCharacterTextSplitter(TextSplitter): "\nfor ", "\nwhile ", "\nwhen ", - "\ncase ", "\nelse ", # Split by the normal type of lines "\n\n", @@ -463,7 +462,6 @@ class RecursiveCharacterTextSplitter(TextSplitter): "\nfor ", "\nloop ", "\nmatch ", - "\nconst ", # Split by the normal type of lines "\n\n", "\n", @@ -718,7 +716,6 @@ class RecursiveCharacterTextSplitter(TextSplitter): "\ndata ", "\nnewtype ", "\ntype ", - "\n:: ", # Split along module declarations "\nmodule ", # Split along import statements @@ -733,7 +730,6 @@ class RecursiveCharacterTextSplitter(TextSplitter): # Split along guards in function definitions "\n| ", # Split along record field declarations - "\ndata ", "\n= {", "\n, ", # Split by the normal type of lines