> ## Documentation Index
> Fetch the complete documentation index at: https://doc.lucidworks.com/llms.txt
> Use this file to discover all available pages before exploring further.

# Chunking Neural Hybrid Query Stage

export const schema = {
  "type": "object",
  "title": "Chunking Neural Hybrid Query",
  "description": "Combines multi-vector similarity search with lexical search to retrieve and re-rank results from chunked documents. Skipped if the query is blank or a wildcard.",
  "required": ["lexicalQuery", "lexicalWeight", "lexicalSquash", "vectorQueryField", "vector", "vectorWeight", "minReturnSim", "minTraverseSim"],
  "properties": {
    "skip": {
      "type": "boolean",
      "title": "Skip This Stage",
      "description": "Controls whether this stage executes during pipeline processing at runtime. When set to `true`, the stage is completely bypassed and documents pass through unchanged to the next stage. Useful for A/B testing, gradual rollouts, or temporarily disabling a stage without removing it from the pipeline.",
      "default": false,
      "hints": ["advanced"]
    },
    "label": {
      "type": "string",
      "title": "Label",
      "description": "Human-readable identifier displayed in the Fusion Admin UI, monitoring dashboards, and log messages. Use descriptive labels like `Parse Product PDFs` to aid debugging and team collaboration. Labels appear in performance metrics and error reports, making it easier to identify which stage failed.",
      "hints": ["advanced"],
      "maxLength": 255
    },
    "condition": {
      "type": "string",
      "title": "Condition",
      "description": "JavaScript expression evaluated at runtime on each document to conditionally execute this stage. The expression must return `true` to execute or `false` to skip. Access document fields using `doc.getFieldValue('fieldName')` and request parameters via `request.getFirstParam('paramName')`. For example, `doc.getFieldValue('type') === 'premium'` executes this stage only for premium content.",
      "hints": ["code", "code/javascript", "advanced"]
    },
    "legacy": {
      "type": "boolean",
      "title": "Legacy",
      "description": "When `true`, this stage operates in legacy mode only.",
      "hints": ["readonly", "hidden"]
    },
    "lexicalQuery": {
      "type": "string",
      "title": "Lexical Query Input",
      "description": "Specifies the source of the lexical query. Supports template expressions such as `<request.params.q>` to use the original user query.",
      "default": "<request.params.q>"
    },
    "lexicalWeight": {
      "type": "number",
      "title": "Lexical Query Weight",
      "description": "Sets the relative weight of lexical results in re-ranking. Set to `0` to disable lexical re-ranking.",
      "default": 0.1,
      "maximum": 10,
      "exclusiveMaximum": false
    },
    "lexicalSquash": {
      "type": "number",
      "title": "Lexical Query Squash Factor",
      "description": "Sets the squash factor applied to lexical scores, converting them from `0..inf` to `0..1` to prevent lexical results from dominating the final score.",
      "default": 0.1,
      "maximum": 10,
      "exclusiveMaximum": false
    },
    "vectorQueryField": {
      "type": "string",
      "title": "Vector Query Field",
      "description": "Specifies the vector field name used for vector similarity search in the chunked document collection."
    },
    "vector": {
      "type": "string",
      "title": "Vector Input",
      "description": "Specifies the source of the query vector. Supports template expressions such as `<ctx.vector>` to reference a context variable from a preceding Vectorize stage.",
      "default": "<ctx.vector>"
    },
    "vectorWeight": {
      "type": "number",
      "title": "Vector Query Weight",
      "description": "Sets the relative weight of vector results in the final re-ranked score.",
      "default": 0.9,
      "maximum": 10,
      "exclusiveMaximum": false,
      "minimum": 0.001,
      "exclusiveMinimum": false
    },
    "minReturnSim": {
      "type": "number",
      "title": "Min Return Vector Similarity",
      "description": "Sets the minimum vector similarity threshold for a document to qualify as a match from the vector portion of the query.",
      "default": 0.5,
      "maximum": 1,
      "exclusiveMaximum": false
    },
    "minTraverseSim": {
      "type": "number",
      "title": "Min Traversal Vector Similarity",
      "description": "Sets the minimum vector similarity used when traversing the graph during the vector search phase. Must be less than or equal to `minReturnSim`.",
      "default": 0.5,
      "maximum": 1,
      "exclusiveMaximum": false
    },
    "vecSimForLexOnly": {
      "type": "boolean",
      "title": "Compute Vector Similarity for Lexical-Only Matches",
      "description": "Controls whether vector similarity scores are computed for documents that appear in the lexical result set but not in the vector result set.",
      "default": true
    },
    "vecPreFilterBoolean": {
      "type": "boolean",
      "title": "Block pre-filtering.",
      "description": "When enabled, adds `preFilter=\"\"` to the vector query to disable Solr pre-filtering during vector search.",
      "default": true
    }
  },
  "category": "AI",
  "categoryPriority": 10,
  "unsafe": false
};

export const SchemaParamFields = ({schema}) => {
  const sanitize = str => {
    if (typeof str !== "string") return str;
    return str.replace(/^"(.*)"$/s, "$1").replace(/\\/g, "").replace(/"/g, "'");
  };
  const renderMd = str => {
    const s = sanitize(str);
    const text = (/[.!?]\)*$/).test(s) ? s : `${s}.`;
    return text.split(/(\*\*[^*]+\*\*|_[^_]+_|`[^`]+`)/g).map((part, i) => {
      if (part.startsWith("**")) return <strong key={i}>{part.slice(2, -2)}</strong>;
      if (part.startsWith("_")) return <em key={i}>{part.slice(1, -1)}</em>;
      if (part.startsWith("`")) return <code key={i}>{part.slice(1, -1)}</code>;
      return part;
    });
  };
  const {description, properties = {}, required: requiredProps = []} = schema;
  const visibleProps = useMemo(() => Object.entries(properties).filter(([, prop]) => !prop.hints?.includes("hidden")), [properties]);
  const renderProp = ([name, prop]) => {
    const isRequired = requiredProps.includes(name);
    const hasDefault = prop.default !== undefined;
    const rawDefault = prop.default;
    const hints = prop.hints || [];
    const isComplexDefault = hasDefault && (typeof rawDefault === "object" || typeof rawDefault === "string" && (rawDefault.length > 20 || rawDefault.includes('"')));
    const postBadges = [];
    if (prop.title) {
      postBadges.push(<><span className="text-stone-400 dark:text-stone-500">API property: </span>{name}</>);
    }
    const constraints = [];
    if (prop.minimum !== undefined && prop.maximum !== undefined) {
      constraints.push(`Range: ${prop.minimum} – ${prop.maximum}`);
    } else if (prop.minimum !== undefined) {
      constraints.push(`Min: ${prop.minimum}`);
    } else if (prop.maximum !== undefined) {
      constraints.push(`Max: ${prop.maximum}`);
    }
    if (prop.minLength !== undefined && prop.maxLength !== undefined) {
      constraints.push(`Length: ${prop.minLength} – ${prop.maxLength}`);
    } else if (prop.minLength !== undefined) {
      constraints.push(`Min length: ${prop.minLength}`);
    } else if (prop.maxLength !== undefined) {
      constraints.push(`Max length: ${prop.maxLength}`);
    }
    const fieldProps = {
      key: name,
      body: prop.title || name,
      type: prop.type,
      ...postBadges.length > 0 && ({
        post: postBadges
      }),
      ...isRequired && ({
        required: true
      }),
      ...!isComplexDefault && hasDefault ? {
        default: sanitize(String(rawDefault))
      } : {}
    };
    const isObject = prop.type === "object" && prop.properties;
    const isArrayOfObjects = prop.type === "array" && prop.items?.type === "object" && prop.items.properties;
    return <ParamField {...fieldProps}>
        {prop.description && <p>{renderMd(prop.description)}</p>}

        {prop.enum && <p>
            Allowed values: 
            {prop.enum.map((v, i) => <>{i > 0 && ", "}<code key={i}>{String(v)}</code></>)}
          </p>}

        {constraints.length > 0 && <p className="text-stone-500 dark:text-stone-400 text-sm">
            {constraints.join(" · ")}
          </p>}

        {isComplexDefault && <div className="flex">
            <p>
              <strong>Default:</strong>
            </p>
            <pre className="!my-0">
              <code>
                {JSON.stringify(rawDefault, null, 2)}
              </code>
            </pre>
          </div>}

        {isArrayOfObjects && <Expandable title="item properties">
            <SchemaParamFields schema={{
      properties: prop.items.properties,
      required: prop.items.required
    }} />
          </Expandable>}

        {isObject && <Expandable title="properties">
            <SchemaParamFields schema={{
      properties: prop.properties,
      required: prop.required
    }} />
          </Expandable>}
      </ParamField>;
  };
  return <div>
      {description && <p>{renderMd(description)}</p>}

      {visibleProps.map(renderProp)}
    </div>;
};

export const LwTemplate = ({title = "Key questions to get you started", icon = "sparkles", cta = "Powered by Agent Studio", linkHref = "https://lucidworks.com/demo/?utm_source=docs&utm_medium=referral&utm_campaign=docs_cta_ai"}) => {
  const [isLoaded, setIsLoaded] = useState(false);
  useEffect(() => {
    const timer = setTimeout(() => {
      setIsLoaded(true);
    }, 500);
    return () => clearTimeout(timer);
  }, []);
  return <div className="lw-template-container">
      <Card title={title} icon={icon}>
        {isLoaded && <span dangerouslySetInnerHTML={{
    __html: `<lw-template id="a029c1a9-28be-427e-b0e1-5d918920246a"></lw-template
            >`
  }} />}
        <Link href={linkHref} className="agent-studio-link text-left text-gray-600 gap-2 dark:text-gray-400 text-sm font-medium flex flex-row items-center hover:text-primary dark:hover:text-primary-light group-hover:text-primary group-hover:dark:text-primary-light">Powered by Lucidworks Agent Studio</Link>
      </Card>
    </div>;
};

[localhost link]: http://localhost:3000/docs/5/fusion/reference/config-ref/pipeline-stages/query-stages/chunking-neural-hybrid-query

[mintlify link]: https://doc.lucidworks.com/docs/5/fusion/reference/config-ref/pipeline-stages/query-stages/chunking-neural-hybrid-query

[old doc.lw link]: https://doc.lucidworks.com/fusion/5.9/4klbo6

Fusion 5.9.12 and later releases use index and query stages to split large documents into smaller, more manageable segments called chunks. For more information about chunking, chunking strategies and setting up chunking, see [Chunking](/docs/5/fusion/hybrid-search/chunking).

The Chunking Neural Hybrid Query stage performs hybrid lexical-semantic search that combines BM25-type lexical search with KNN dense vector search via Solr. This stage differs from the [Neural Hybrid Stage](/docs/5/fusion/reference/config-ref/pipeline-stages/query-stages/neural-hybrid-query) because it supports chunking.

Not sure which hybrid query stage is right for you? Read about the
[differences between the hybrid query stages](/docs/5/fusion/hybrid-search/hybrid-stage-differences).

<Note>
  This feature is available in Fusion 5.9.12 and later.

  Some prefiltering capabilities, such as access to the `preFilterKey` context property and the `VectorPreFilter` helper class are available in 5.9.13 and later.
</Note>

Click **Get Started** below to see how to enable chunking in Fusion:

<iframe src="https://app.supademo.com/embed/cmfzg6uw4009oxx0i1ptmac82?embed_v=2&utm_source=embed" loading="lazy" title="Enable chunking in Fusion" allow="clipboard-write" frameborder="0" webkitallowfullscreen="true" mozallowfullscreen="true" allowfullscreen style={{  width: '100%', height: '500px' }} />

<LwTemplate />

Click your use case below to see examples of how the Chunking Neural Hybrid Query Stage processes documents:

<Tabs>
  <Tab title="B2B inventory management">
    In a single document, a manufacturing company indexes its parts catalog, where each record contains product specifications, compatibility notes, reorder thresholds, and supplier lead times.

    A procurement manager queries: *"which hydraulic fittings are compatible with the Model 7 assembly line?"*

    Without chunking, the full parts record competes as one unit and the compatibility section is diluted by pricing and supplier content that has no relevance to the query.

    With chunking, Fusion isolates the compatibility paragraph as its own chunk and scores it independently, returning the most relevant part records at the top even when the rest of the document covers unrelated logistics details.

    This pattern is common in B2B catalog search where product records are dense and structured around operational concerns. A single SKU might describe assembly specs, regulatory certifications, hazmat handling, and pricing in the same document.

    Chunking lets users find the specific detail they need without requiring the entire record to be relevant.
  </Tab>

  <Tab title="B2C product discovery">
    An e-commerce retailer indexes product pages that each contain a marketing description, technical specifications, customer reviews, and care instructions.

    A shopper queries: *"outdoor jacket that's waterproof but also machine washable"*.

    The waterproof detail is in the technical specs chunk, but the machine washable detail is in a separate section in the care instructions chunk.

    The Chunking Neural Hybrid Query stage scores each chunk individually and uses the best-matching chunk's vector score to represent the parent document.

    That top chunk's vector score then combines with the whole-document lexical score, which sees both terms across the full page, so the parent product surfaces in results even though no single chunk contains both terms together.

    For B2C search this is particularly valuable for long-form product pages where user intent can match different parts of the same document simultaneously.

    Chunking surfaces those matches instead of burying them in an aggregate score dominated by the marketing copy.
  </Tab>

  <Tab title="Knowledge management: HR family leave policy">
    In one long document, an HR team indexes its employee handbook, which includes the family leave policy that covers eligibility criteria, benefit duration, the leave request procedure, manager responsibilities, and return-to-work steps.

    An employee queries: *"how do I request family leave if I've been employed less than a year?"*

    Without chunking, the full policy document scores as a single unit and the eligibility chunk, which directly answers the question, is weighted equally with sections about manager approval workflows that are irrelevant to the employee's intent.

    With chunking, the eligibility and request procedure chunks rank first, surfacing the directly actionable answer.

    This applies broadly to knowledge management use cases where policy documents, runbooks, and procedures pack multiple distinct procedures into a single file.

    Chunking ensures that a question about one procedure doesn't require the entire document to be topically aligned.
  </Tab>
</Tabs>

## About the Lexical Query Squash Factor

The **Lexical Query Squash Factor** field lets you input a value that squashes the lexical query scores from `0..inf` to `0..1`.
This setting helps prevent the lexical query from dominating the final score, and normalizes the score into a range that works well with vector similarity scores.
Additionally, it helps prevent the [vanishing gradient problem](https://en.wikipedia.org/wiki/Vanishing_gradient_problem), which occurs when very high lexical scores are mapped to values extremely close to `1`, such as `0.99999999`.
During the hybrid search calculation, these near-1 values can cause the system to lose sensitivity to subtle differences in lexical relevance, effectively 'squashing' the gradient and reducing the impact of lexical scoring.

Lucidworks recommends setting the **Lexical Query Squash Factor** to the inverse of the maximum lexical score observed across your queries.
This helps balance the impact of lexical and vector scores, leading to more accurate and nuanced search results.

## Prefiltering

Prefiltering is a technique that can improve performance and accuracy by filtering documents before applying the algorithm, reducing the number of documents that need to be processed.
This is especially effective with the KNN algorithm.

Prefiltering is disabled by default.
**To enable it, uncheck **Block pre-filtering** in this stage.**

When prefiltering is enabled, you can configure the filters using one or both of these methods:

* **Security filters**\
  You can use security filters as prefilters by placing the [Graph Security Trimming Stage](/docs/5/fusion/reference/config-ref/pipeline-stages/query-stages/security-trimming-graph-query-stage) *after* this one in the pipeline.\
  Then Fusion uses the security trimming filter as a prefilter.

* **JavaScript**\
  When prefiltering is enabled, this stage adds a `preFilterKey` object to the Javascript `ctx` object.\
  You can place a [Javascript stage](/docs/5/fusion/reference/config-ref/pipeline-stages/query-stages/javascript-query-stage) after this one and use it to access the `preFilterKey` object, as in this example:

  ```js theme={"dark"}
  if(ctx.hasProperty("preFilterKey")) {
    var preFilter = ctx.getProperty("preFilterKey");
    preFilter.addFilter(filterQuery)
  }
  ```

  You can also use the following example in Fusion 5.9.13 and later for placing pre-filter specific filters:

  ```js theme={"dark"}
  var QueryRequestAndResponse = Java.type('com.lucidworks.apollo.pipeline.query.QueryRequestAndResponse');
  var VectorPreFilter = Java.type("com.lucidworks.apollo.pipeline.query.stages.VectorPreFilter");
  var preFilter = ctx.get(VectorPreFilter.CONTEXT_KEY);
  if(preFilter){
    var wrapper = QueryRequestAndResponse.create(request,response,0)
      preFilter.addFilter(wrapper, 'id:* OR foo_s:bar');
  }
  ```

  <Note>
    The context object of the `VectorPreFilter` class is not present in Fusion 5.9.12 and earlier. Use the Additional Query Parameters stage example and the 5.9.12 note below.
  </Note>

  In Fusion 5.9.12 and earlier, a non-JavaScript approach is required in addition to the Additional Query Parameters stage example in the next section.

  In Fusion 5.9.12 the parameter `vec_sim_q` after the Chunking Neural Hybrid Query stage needs to be altered to include `{!knn f=$vec_field v=$vec_q topK=100 preFilter=$vectorPreFilter}` where `topK` is consistent with what your stage sets.

  ```js theme={"dark"}
  if (request.hasParam("vec_sim_q")) {
      var vecSimQ = request.getFirstFieldValue("vec_sim_q");
      if (vecSimQ.indexOf("preFilter") < 0) {
        vecSimQ = vecSimQ.substring(0,vecSimQ.lastIndexOf("}"))  + " preFilter=$vectorPreFilter }";
        request.removeParam("vec_sim_q")
        request.addParam("vec_sim_q",vecSimQ);
      }
    }
  ```

* **Additional Query Parameters stage**

  If you do not want to create a JavaScript stage, you can create additional query parameters to prefilter the documents to be processed by using what the previous JavaScript example adds to the request. This step is required for Fusion 5.9.12. The following example uses a single prefilter:

  ```js theme={"dark"}
  "fq" = "{!bool filter=$vectorPreFilter}"
  "vectorPreFilter" = "EXAMPLE_FILTER"
  ```

  The following example uses multiple prefilters:

  ```js theme={"dark"}
  "fq": "{!bool filter=$filterClauses}",
  "vectorPreFilter": "{!bool should=$filterClauses}",
  "filterClauses": ["id:EXAMPLE_FILTER1","id:EXAMPLE_FILTER2"]
  ```

## Query pipeline stage condition examples

Stages can be triggered conditionally when a script in the **Condition** field evaluates to true.
Some examples are shown below.

Run this stage only for mobile clients:

```js wrap  theme={"dark"}
params.deviceType === "mobile"
```

Run this stage when debugging is enabled:

```js wrap  theme={"dark"}
params.debug === "true"
```

Run this stage when the query includes a specific term:

```js wrap  theme={"dark"}
params.q && params.q.includes("sale")
```

Run this stage when multiple conditions are met:

```js wrap  theme={"dark"}
request.hasParam("fusion-user-name") && request.getFirstParam("fusion-user-name").equals("SuperUser");
!request.hasParam("isFusionPluginQuery")
```

The first condition checks that the request parameter "fusion-user-name" is present and has the value "SuperUser".
The second condition checks that the request parameter "isFusionPluginQuery" is not present.

## Configuration

<Tip>
  When entering configuration values in the UI, use *unescaped* characters, such as `\t` for the tab character. When entering configuration values in the API, use *escaped* characters, such as `\\t` for the tab character.
</Tip>

<SchemaParamFields schema={schema} />
