# Issue while indexing with custom analyzers

**URL:** <https://discuss.elastic.co/t/issue-while-indexing-with-custom-analyzers/363265>\
**Category:** Elasticsearch\
**Created:** [July 17, 2024, 10:00am UTC](https://discuss.elastic.co/t/issue-while-indexing-with-custom-analyzers/363265 "2024-07-17T10:00:19Z")\
**Posts on this page:** 4\
**Page:** 1

<div class="post-metadata">

**Author:** ![NSGokul](https://avatars.discourse-cdn.com/v4/letter/n/f0a364/32.png) [@NSGokul](https://discuss.elastic.co/u/NSGokul)\
**Post date:** [July 17, 2024, 10:00am UTC](https://discuss.elastic.co/t/issue-while-indexing-with-custom-analyzers/363265/1 "2024-07-17T10:00:19Z")

</div>

I've created an Elasticsearch index with custom settings and analyzers to handle various text transformations. Here are the settings for my index:

```auto
"settings": {
  "index": {
    "analysis": {
      "filter": {
        "my_synonyms_2_first_name": {
          "type": "synonym",
          "synonyms_path": "Broadness_3_First_Name.txt"
        },
        "my_synonyms_first_name": {
          "type": "synonym",
          "synonyms_path": "Broadness_2_First_Name.txt"
        },
        "french_stop": {
          "type": "stop",
          "stopwords": "_french_"
        },
        "english_stemmer": {
          "name": "english",
          "type": "stemmer"
        },
        "my_english_soundex": {
          "replace": "false",
          "type": "phonetic",
          "encoder": "soundex"
        },
        "my_french_soundex": {
          "languageset": ["french"],
          "replace": "false",
          "rule_type": "approx",
          "type": "phonetic",
          "encoder": "beider_morse",
          "name_type": "generic"
        },
        "my_synonyms_last_name": {
          "type": "synonym",
          "synonyms_path": "Broadness_2_Last_Name.txt"
        },
        "english_stop": {
          "type": "stop",
          "stopwords": "_english_"
        },
        "french_elision": {
          "type": "elision",
          "articles": [
            "l", "m", "t", "qu", "n", "s", "j", "d", "c", "jusqu", "quoiqu", "lorsqu", "puisqu"
          ],
          "articles_case": "true"
        },
        "my_location_synonyms": {
          "type": "synonym",
          "synonyms_path": "LOCATION.txt"
        },
        "my_synonyms_2_last_name": {
          "type": "synonym",
          "synonyms_path": "Broadness_3_Last_Name.txt"
        },
        "french_stemmer": {
          "name": "french",
          "type": "stemmer"
        }
      },
      "analyzer": {
        "uppercase": {
          "filter": ["uppercase", "asciifolding"],
          "tokenizer": "standard"
        },
        "location_synonyms": {
          "filter": ["lowercase", "asciifolding", "my_location_synonyms"],
          "tokenizer": "keyword"
        },
        "my_synonyms": {
          "filter": ["lowercase", "asciifolding", "my_synonyms_first_name", "my_synonyms_last_name"],
          "tokenizer": "standard"
        },
        "lowercase": {
          "filter": ["lowercase", "asciifolding"],
          "tokenizer": "standard"
        },
        "my_english_phonetic": {
          "filter": ["lowercase", "my_english_soundex", "asciifolding"],
          "tokenizer": "standard"
        },
        "my_synonyms_2": {
          "filter": ["lowercase", "asciifolding", "my_synonyms_2_first_name", "my_synonyms_2_last_name"],
          "tokenizer": "standard"
        },
        "english": {
          "filter": ["lowercase", "asciifolding"],
          "tokenizer": "standard"
        },
        "fuzzy_analyzer": {
          "filter": ["lowercase"],
          "tokenizer": "standard"
        },
        "my_french_phonetic": {
          "filter": ["lowercase", "french_stop", "french_stemmer", "my_french_soundex"],
          "tokenizer": "standard"
        },
        "french": {
          "filter": ["lowercase", "french_stop", "french_stemmer", "french_elision"],
          "tokenizer": "standard"
        }
      }
    },
    "similarity": {
      "scripted_tfidf": {
        "type": "scripted",
        "script": {
          "source": "double tf = Math.sqrt(doc.freq); double idf = 1.0; double norm = 1/Math.sqrt(doc.length); return query.boost * tf * idf * norm;"
        }
      }
    }
  }
}

```

I have also created multiple field types for each text field with these custom analyzers. Here is an example for the `prenom_conjointe_precedente_du_probant` field:

```auto
"prenom_conjointe_precedente_du_probant": {
  "similarity": "scripted_tfidf",
  "type": "text",
  "fields": {
    "uppercase": {
      "analyzer": "uppercase",
      "similarity": "scripted_tfidf",
      "type": "text"
    },
    "english_phonetic": {
      "analyzer": "my_english_phonetic",
      "similarity": "scripted_tfidf",
      "type": "text"
    },
    "exact_match": {
      "similarity": "scripted_tfidf",
      "type": "keyword"
    },
    "french_phonetic": {
      "analyzer": "my_french_phonetic",
      "similarity": "scripted_tfidf",
      "type": "text"
    },
    "lowercase": {
      "analyzer": "lowercase",
      "similarity": "scripted_tfidf",
      "type": "text"
    },
    "english": {
      "analyzer": "english",
      "similarity": "scripted_tfidf",
      "type": "text"
    },
    "broadness_3_synonyms": {
      "analyzer": "my_synonyms_2",
      "similarity": "scripted_tfidf",
      "type": "text"
    },
    "suggest": {
      "max_input_length": 50,
      "analyzer": "simple",
      "preserve_position_increments": true,
      "type": "completion",
      "preserve_separators": true
    },
    "broadness_2_synonyms": {
      "analyzer": "my_synonyms",
      "similarity": "scripted_tfidf",
      "type": "text"
    },
    "french": {
      "analyzer": "french",
      "similarity": "scripted_tfidf",
      "type": "text"
    },
    "fuzzy": {
      "analyzer": "fuzzy_analyzer",
      "type": "text"
    }
  },
  "copy_to": [
    "full_name_conjointe_precedente_du_probant"
  ]
}

```

After indexing the data, I noticed that the fields using the `lowercase` analyzer did not transform the data to lowercase or remove ASCII characters as expected.

![Screenshot from 2024-07-17 15-27-40](https://us1.discourse-cdn.com/elastic/original/3X/f/d/fda1437a36a77317f7ce647fa2ae633533a65986.png)

Could someone help me understand why the `lowercase` analyzer is not applying the expected transformations on my indexed data? Is there something I'm missing in the configuration or the way data is being indexed? Any insights or solutions would be greatly appreciated!

---

<div class="post-metadata">

**Author:** ![dadoonet](https://sea2.discourse-cdn.com/elastic/user_avatar/discuss.elastic.co/dadoonet/32/137187_2.png) [@dadoonet](https://discuss.elastic.co/u/dadoonet)\
**Post date:** [July 17, 2024, 11:30am UTC](https://discuss.elastic.co/t/issue-while-indexing-with-custom-analyzers/363265/2 "2024-07-17T11:30:44Z")

</div>

I can't see the mapping for the lowercase field. What is it?  
If you did not define it, most likely you are using the standard analyzer.

---

<div class="post-metadata">

**Author:** ![NSGokul](https://avatars.discourse-cdn.com/v4/letter/n/f0a364/32.png) [@NSGokul](https://discuss.elastic.co/u/NSGokul)\
**Post date:** [July 17, 2024, 12:09pm UTC](https://discuss.elastic.co/t/issue-while-indexing-with-custom-analyzers/363265/4 "2024-07-17T12:09:04Z")

</div>

@ [dadoonet](https://discuss.elastic.co/u/dadoonet) could you please scroll down the field types of prenom\_conjointe\_precedente\_du\_probant, there you can see the field type with lowercase analyzer

---

<div class="post-metadata">

**Author:** ![dadoonet](https://sea2.discourse-cdn.com/elastic/user_avatar/discuss.elastic.co/dadoonet/32/137187_2.png) [@dadoonet](https://discuss.elastic.co/u/dadoonet)\
**Post date:** [July 17, 2024, 1:10pm UTC](https://discuss.elastic.co/t/issue-while-indexing-with-custom-analyzers/363265/5 "2024-07-17T13:10:52Z")

</div>

In the screen capture there's a nom\_usuel\_du\_probant.lowercase. What is that?

Anyway, could you provide a full recreation script as described in [About the Elasticsearch category](https://discuss.elastic.co/t/about-the-elasticsearch-category/21). It will help to better understand what you are doing. Please, try to keep the example as simple as possible.

A full reproduction script is something anyone can **copy and paste in Kibana dev console** , click on the run button to reproduce your use case. It will help readers to understand, reproduce and if needed fix your problem. It will also most likely help to get a faster answer.

Have a look at the [Elastic Stack and Solutions Help · Forums and Slack | Elastic](https://elastic.co/community/help) page. It contains also lot of useful information on how to ask for help.

Something with only one field would help.
