{ "$schema": "https://json-schema.org/draft/2020-12/schema", "$id": "https://raw.githubusercontent.com/api-evangelist/algolia/main/json-schema/algolia-configuration-schema.json", "title": "Configuration", "description": "Crawler configuration.", "x-generated": "2026-09-23", "x-method": "derived", "x-generator": "derive-json-schema.py", "x-source": "openapi/algolia-crawler-api-openapi.yml#/components/schemas/Configuration", "type": "object", "required": [ "appId", "rateLimit", "actions" ], "properties": { "actions": { "type": "array", "description": "A list of actions.", "minItems": 1, "maxItems": 30, "items": { "$ref": "#/$defs/Action" } }, "apiKey": { "type": "string", "description": "The Algolia API key the crawler uses for indexing records.\nIf you don't provide an API key, one will be generated by the Crawler when you create a configuration.\n\nThe API key must have:\n\n- These [rights and restrictions](https://www.algolia.com/doc/guides/security/api-keys/#rights-and-restrictions): `search`, `addObject`, `deleteObject`, `deleteIndex`, `settings`, `editSettings`, `listIndexes`, `browse`\n- Access to the correct set of indices, based on the crawler's `indexPrefix`.\nFor example, if the prefix is `crawler_`, the API key must have access to `crawler_*`.\n\n**Don't use your [Admin API key](https://www.algolia.com/doc/guides/security/api-keys/#predefined-api-keys)**.\n" }, "appId": { "$ref": "#/$defs/applicationID" }, "exclusionPatterns": { "type": "array", "description": "URLs to exclude from crawling.", "maxItems": 100, "items": { "type": "string", "description": "Use [micromatch](https://github.com/micromatch/micromatch) for negation, wildcards, and more.\n" } }, "externalData": { "type": "array", "description": "References to external data sources for enriching the extracted records.\n", "maxItems": 10, "items": { "type": "string", "description": "For more information, see [Enrich extracted records with external data](https://www.algolia.com/doc/tools/crawler/guides/enriching-extraction-with-external-data)." } }, "extraUrls": { "type": "array", "maxItems": 9999, "description": "The Crawler treats `extraUrls` the same as `startUrls`.\nSpecify `extraUrls` if you want to differentiate between URLs you manually added to fix site crawling from those you initially specified in `startUrls`.\n", "items": { "type": "string" } }, "ignoreCanonicalTo": { "$ref": "#/$defs/ignoreCanonicalTo" }, "ignoreNoFollowTo": { "type": "boolean", "description": "Determines if the crawler should follow links with a `nofollow` directive.\nIf `true`, the crawler will ignore the `nofollow` directive and crawl links on the page.\n\nThe crawler always ignores links that don't match your [configuration settings](https://www.algolia.com/doc/tools/crawler/getting-started/crawler-configuration/#exclude-and-include-content).\n`ignoreNoFollowTo` applies to:\n\n- Links that are ignored because the [`robots` meta tag](https://developer.mozilla.org/en-US/docs/Web/HTML/Element/meta/name#Other_metadata_names) contains `nofollow` or `none`.\n- Links with a [`rel` attribute](https://developer.mozilla.org/en-US/docs/Web/HTML/Attributes/rel) containing the `nofollow` directive.\n" }, "ignoreNoIndex": { "type": "boolean", "description": "Whether to ignore the `noindex` robots meta tag.\nIf `true`, pages with this meta tag _will_ be crawled.\n" }, "ignorePaginationAttributes": { "type": "boolean", "description": "Whether the crawler should follow `rel=\"prev\"` and `rel=\"next\"` pagination links in the `` section of an HTML page.\n\n- If `true`, the crawler ignores the pagination links.\n- If `false`, the crawler follows the pagination links.\n", "default": true }, "ignoreQueryParams": { "type": "array", "description": "Query parameters to ignore while crawling.\n\nAll URLs with the matching query parameters are treated as identical.\nThis prevents indexing URLs that just differ by their query parameters.\n", "maxItems": 9999, "items": { "type": "string", "description": "Use wildcards to match multiple query parameters." } }, "ignoreRobotsTxtRules": { "type": "boolean", "description": "Whether to ignore rules defined in your `robots.txt` file." }, "indexPrefix": { "type": "string", "description": "A prefix for all indices created by this crawler. It's combined with the `indexName` for each action to form the complete index name.", "maxLength": 64 }, "initialIndexSettings": { "type": "object", "description": "Crawler index settings.\n\nThese index settings are only applied during the first crawl of an index.\n\nAny subsequent changes won't be applied to the index.\nInstead, make changes to your index settings in the [Algolia dashboard](https://dashboard.algolia.com/explorer/configuration).\n", "additionalProperties": { "$ref": "#/$defs/indexSettings", "x-additionalPropertiesName": "indexName" } }, "linkExtractor": { "title": "linkExtractor", "type": "object", "description": "Function for extracting URLs from links on crawled pages.\n\nFor more information, see the [`linkExtractor` documentation](https://www.algolia.com/doc/tools/crawler/apis/configuration/link-extractor).\n", "properties": { "__type": { "$ref": "#/$defs/configurationRecordExtractorType" }, "source": { "type": "string" } } }, "login": { "$ref": "#/$defs/login" }, "maxDepth": { "type": "integer", "description": "Determines the maximum path depth of crawled URLs.\n\nPath depth is calculated based on the number of slash characters (`/`) after the domain (starting at 1).\nFor example:\n\n- **1** `http://example.com`\n- **1** `http://example.com/`\n- **1** `http://example.com/foo`\n- **2** `http://example.com/foo/`\n- **2** `http://example.com/foo/bar`\n- **3** `http://example.com/foo/bar/`\n\n**URLs added with `startUrls` and `sitemaps` aren't checked for `maxDepth`.**.\n", "minimum": 1, "maximum": 100 }, "maxUrls": { "type": "integer", "description": "Limits the number of URLs your crawler processes.\n\nChange it to a low value, such as 100, for short crawling tests.\nChange it to a higher explicit value for full crawls to prevent it from getting \"lost\" in complex site structures.\nBecause the Crawler works on many pages simultaneously, `maxUrls` doesn't guarantee finding the same pages each time it runs.\n", "minimum": 1, "maximum": 15000000 }, "rateLimit": { "type": "integer", "description": "Determines the number of concurrent tasks per second that can run for this configuration.\n\nA higher rate limit means more crawls per second.\nAlgolia prevents system overload by ensuring the number of URLs added in the last second and the number of URLs being processed is less than the rate limit:\n\n```\nmax(new_urls_added, active_urls_processing) <= rateLimit\n```\n\nStart with a low value (for example, 2) and increase it if you need faster crawling.\nA high `rateLimit` can significantly increase bandwidth cost and server resource consumption.\n\nThe number of pages processed per second depends on the average time it takes to fetch, process, and upload a URL.\nFor a given `rateLimit`, if fetching, processing, and uploading URLs takes (on average):\n\n- Less than a second, your crawler processes up to `rateLimit` pages per second.\n- Four seconds, your crawler processes up to `rateLimit / 4` pages per second.\n\nIn the latter case, increasing `rateLimit` improves performance up to a point.\nIf the processing time remains at four seconds, increasing `rateLimit` won't increase the number of pages processed per second.\n", "minimum": 1, "maximum": 100 }, "renderJavaScript": { "$ref": "#/$defs/renderJavaScript" }, "requestOptions": { "$ref": "#/$defs/requestOptions" }, "safetyChecks": { "$ref": "#/$defs/safetyChecks" }, "saveBackup": { "type": "boolean", "description": "Whether to back up your index before the crawler overwrites it with new records.\n" }, "schedule": { "$ref": "#/$defs/schedule" }, "sitemaps": { "type": "array", "description": "Sitemaps with URLs from where to start crawling.", "maxItems": 9999, "items": { "type": "string" } }, "startUrls": { "type": "array", "description": "URLs from where to start crawling.", "maxItems": 9999, "items": { "type": "string" } } }, "$defs": { "Action": { "type": "object", "description": "How to process crawled URLs.\n\nEach action defines:\n\n- The targeted subset of URLs it processes.\n- What information to extract from the web pages.\n- The Algolia indices where the extracted records will be stored.\n\nIf a single web page matches several actions,\none record is generated for each action.\n", "properties": { "autoGenerateObjectIDs": { "type": "boolean", "description": "Whether to generate an `objectID` for records that don't have one.", "default": true }, "cache": { "$ref": "#/$defs/cache" }, "discoveryPatterns": { "type": "array", "description": "Which _intermediary_ web pages the crawler should visit.\nUse `discoveryPatterns` to define pages that should be visited _just_ for their links to other pages,\n_not_ their content.\nIt functions similarly to the `pathsToMatch` action but without record extraction.\n", "items": { "$ref": "#/$defs/urlPattern" } }, "fileTypesToMatch": { "type": "array", "description": "File types for crawling non-HTML documents.\n", "maxItems": 100, "items": { "$ref": "#/$defs/fileTypes" }, "default": [ "html" ] }, "hostnameAliases": { "$ref": "#/$defs/hostnameAliases" }, "indexName": { "type": "string", "maxLength": 256, "description": "Reference to the index used to store the action's extracted records.\n`indexName` is combined with the prefix you specified in `indexPrefix`.\n" }, "name": { "type": "string", "description": "Unique identifier for the action. This option is required if `schedule` is set." }, "pathAliases": { "$ref": "#/$defs/pathAliases" }, "pathsToMatch": { "type": "array", "description": "URLs to which this action should apply.\n\nUses [micromatch](https://github.com/micromatch/micromatch) for negation, wildcards, and more.\n", "minItems": 1, "maxItems": 100, "items": { "$ref": "#/$defs/urlPattern" } }, "recordExtractor": { "title": "recordExtractor", "type": "object", "description": "Function for extracting information from a crawled page and transforming it into Algolia records for indexing.\n\nThe Crawler has an [editor](https://www.algolia.com/doc/tools/crawler/getting-started/crawler-configuration/#the-editor) with autocomplete and validation to help you update the `recordExtractor`.\nFor details, see the [`recordExtractor` documentation](https://www.algolia.com/doc/tools/crawler/apis/configuration/actions/#parameter-param-recordextractor).\n", "properties": { "__type": { "$ref": "#/$defs/configurationRecordExtractorType" }, "source": { "type": "string", "description": "A JavaScript function (as a string) that returns one or more Algolia records for each crawled page.\n" } } }, "schedule": { "type": "string", "description": "How often to perform a complete crawl for this action.\n\nFor mopre information, consult the [`schedule` parameter documentation](https://www.algolia.com/doc/tools/crawler/apis/configuration/schedule).\n" }, "selectorsToMatch": { "type": "array", "description": "DOM selectors for nodes that must be present on the page to be processed.\nIf the page doesn't match any of the selectors, it's ignored.\n", "maxItems": 100, "items": { "type": "string", "description": "Prefix a selector with `!` to ignore matching pages. \n" } } }, "required": [ "indexName", "recordExtractor" ] }, "IndexSettings_advancedSyntaxFeatures": { "type": "array", "items": { "$ref": "#/$defs/advancedSyntaxFeatures" }, "description": "Advanced search syntax features you want to support\n- `exactPhrase`.\n Phrases in quotes must match exactly.\n For example, `sparkly blue \"iPhone case\"` only returns records with the exact string \"iPhone case\"\n- `excludeWords`.\n Query words prefixed with a `-` must not occur in a record.\n For example, `search -engine` matches records that contain \"search\" but not \"engine\"\nThis setting only has an effect if `advancedSyntax` is true.\n", "default": [ "exactPhrase", "excludeWords" ], "x-categories": [ "Query strategy" ] }, "IndexSettings_alternativesAsExact": { "type": "array", "items": { "$ref": "#/$defs/alternativesAsExact" }, "description": "Determine which plurals and synonyms should be considered an exact matches\nBy default, Algolia treats singular and plural forms of a word, and single-word synonyms, as [exact](https://www.algolia.com/doc/guides/managing-results/relevance-overview/in-depth/ranking-criteria/#exact) matches when searching.\nFor example\n- \"swimsuit\" and \"swimsuits\" are treated the same\n- \"swimsuit\" and \"swimwear\" are treated the same (if they are [synonyms](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/adding-synonyms/#regular-synonyms))\n- `ignorePlurals`.\n Plurals and similar declensions added by the `ignorePlurals` setting are considered exact matches\n- `singleWordSynonym`.\n Single-word synonyms, such as \"NY\" = \"NYC\", are considered exact matches\n- `multiWordsSynonym`.\n Multi-word synonyms, such as \"NY\" = \"New York\", are considered exact matches.\n", "default": [ "ignorePlurals", "singleWordSynonym" ], "x-categories": [ "Query strategy" ] }, "accessTokenRequest": { "type": "object", "description": "Parameters required to make the [access token request](https://datatracker.ietf.org/doc/html/rfc6749#section-4.4.2).\n", "properties": { "url": { "type": "string", "description": "URL for the access token endpoint." }, "grantType": { "$ref": "#/$defs/grantType" }, "clientId": { "type": "string", "description": "[Client identifier](https://datatracker.ietf.org/doc/html/rfc6749#section-2.2).\n" }, "clientSecret": { "type": "string", "description": "Client secret." }, "scope": { "type": "string", "description": "[Access token scope](https://datatracker.ietf.org/doc/html/rfc6749#section-3.3).\n" }, "extraParameters": { "$ref": "#/$defs/extraParameters" } }, "required": [ "url", "grantType", "clientId", "clientSecret" ] }, "advancedSyntax": { "type": "boolean", "description": "Whether to support phrase matching and excluding words from search queries\nUse the `advancedSyntaxFeatures` parameter to control which feature is supported.\n", "default": false, "x-categories": [ "Query strategy" ] }, "advancedSyntaxFeatures": { "type": "string", "enum": [ "exactPhrase", "excludeWords" ], "x-categories": [ "Query strategy" ] }, "allowTyposOnNumericTokens": { "type": "boolean", "description": "Whether to allow typos on numbers in the search query\nTurn off this setting to reduce the number of irrelevant matches\nwhen searching in large sets of similar numbers.\n", "default": true, "x-categories": [ "Typos" ] }, "alternativesAsExact": { "type": "string", "enum": [ "ignorePlurals", "singleWordSynonym", "multiWordsSynonym", "ignoreConjugations" ], "x-categories": [ "Query strategy" ] }, "applicationID": { "type": "string", "description": "Algolia application ID where the crawler creates and updates indices.\n" }, "attributeCriteriaComputedByMinProximity": { "type": "boolean", "description": "Whether the best matching attribute should be determined by minimum proximity\nThis setting only affects ranking if the Attribute ranking criterion comes before Proximity in the `ranking` setting.\nIf true, the best matching attribute is selected based on the minimum proximity of multiple matches.\nOtherwise, the best matching attribute is determined by the order in the `searchableAttributes` setting.\n", "default": false, "x-categories": [ "Advanced" ] }, "attributesToHighlight": { "type": "array", "items": { "type": "string" }, "description": "Attributes to highlight\nBy default, all searchable attributes are highlighted.\nUse `*` to highlight all attributes or use an empty array `[]` to turn off highlighting.\nAttribute names are case-sensitive\nWith highlighting, strings that match the search query are surrounded by HTML tags defined by `highlightPreTag` and `highlightPostTag`.\nYou can use this to visually highlight matching parts of a search query in your UI\nFor more information, see [Highlighting and snippeting](https://www.algolia.com/doc/guides/building-search-ui/ui-and-ux-patterns/highlighting-snippeting/js).\n", "x-categories": [ "Highlighting and Snippeting" ] }, "attributesToRetrieve": { "type": "array", "items": { "type": "string" }, "description": "Attributes to include in the API response\nTo reduce the size of your response, you can retrieve only some of the attributes.\nAttribute names are case-sensitive\n- `*` retrieves all attributes, except attributes included in the `customRanking` and `unretrievableAttributes` settings.\n- To retrieve all attributes except a specific one, prefix the attribute with a dash and combine it with the `*`: `[\"*\", \"-ATTRIBUTE\"]`.\n- The `objectID` attribute is always included.\n", "default": [ "*" ], "x-categories": [ "Attributes" ] }, "attributesToSnippet": { "type": "array", "items": { "type": "string" }, "description": "Attributes for which to enable snippets.\nAttribute names are case-sensitive\nSnippets provide additional context to matched words.\nIf you enable snippets, they include 10 words, including the matched word.\nThe matched word will also be wrapped by HTML tags for highlighting.\nYou can adjust the number of words with the following notation: `ATTRIBUTE:NUMBER`,\nwhere `NUMBER` is the number of words to be extracted.\n", "default": [], "x-categories": [ "Highlighting and Snippeting" ] }, "banner": { "description": "Banner with image and link to redirect users.", "type": "object", "additionalProperties": false, "properties": { "image": { "$ref": "#/$defs/bannerImage" }, "link": { "$ref": "#/$defs/bannerLink" } } }, "bannerImage": { "description": "Image to show inside a banner.", "type": "object", "additionalProperties": false, "properties": { "urls": { "type": "array", "items": { "$ref": "#/$defs/bannerImageUrl" } }, "title": { "type": "string" } } }, "bannerImageUrl": { "description": "URL for an image to show inside a banner.", "type": "object", "additionalProperties": false, "properties": { "url": { "type": "string" } } }, "bannerLink": { "description": "Link for a banner defined in the Merchandising Studio.", "type": "object", "additionalProperties": false, "properties": { "url": { "type": "string" } } }, "banners": { "description": "Banners defined in the Merchandising Studio for a given search.", "type": "array", "items": { "$ref": "#/$defs/banner" } }, "baseIndexSettings": { "type": "object", "additionalProperties": false, "properties": { "attributesForFaceting": { "type": "array", "items": { "type": "string" }, "description": "Attributes used for [faceting](https://www.algolia.com/doc/guides/managing-results/refine-results/faceting).\n\nFacets are attributes that let you categorize search results.\nThey can be used for filtering search results.\nBy default, no attribute is used for faceting.\nAttribute names are case-sensitive.\n\n**Modifiers**\n\n- `filterOnly(\"ATTRIBUTE\")`.\n Allows the attribute to be used as a filter but doesn't evaluate the facet values.\n\n- `searchable(\"ATTRIBUTE\")`.\n Allows searching for facet values.\n\n- `afterDistinct(\"ATTRIBUTE\")`.\n Evaluates the facet count _after_ deduplication with `distinct`.\n This ensures accurate facet counts.\n You can apply this modifier to searchable facets: `afterDistinct(searchable(ATTRIBUTE))`.\n", "default": [], "x-categories": [ "Faceting" ] }, "replicas": { "type": "array", "items": { "type": "string" }, "description": "Creates [replica indices](https://www.algolia.com/doc/guides/managing-results/refine-results/sorting/in-depth/replicas).\n\nReplicas are copies of a primary index with the same records but different settings, synonyms, or rules.\nIf you want to offer a different ranking or sorting of your search results, you'll use replica indices.\nAll index operations on a primary index are automatically forwarded to its replicas.\nTo add a replica index, you must provide the complete set of replicas to this parameter.\nIf you omit a replica from this list, the replica turns into a regular, standalone index that will no longer be synced with the primary index.\n\n**Modifier**\n\n- `virtual(\"REPLICA\")`.\n Create a virtual replica,\n Virtual replicas don't increase the number of records and are optimized for [Relevant sorting](https://www.algolia.com/doc/guides/managing-results/refine-results/sorting/in-depth/relevant-sort).\n", "default": [], "x-categories": [ "Ranking" ] }, "paginationLimitedTo": { "type": "integer", "description": "Maximum number of search results that can be obtained through pagination.\n\nHigher pagination limits might slow down your search.\nFor pagination limits above 1,000, the sorting of results beyond the 1,000th hit can't be guaranteed.\n", "default": 1000, "maximum": 20000 }, "unretrievableAttributes": { "type": "array", "items": { "type": "string" }, "description": "Attributes that can't be retrieved at query time.\n\nThis can be useful if you want to use an attribute for ranking or to [restrict access](https://www.algolia.com/doc/guides/security/api-keys/how-to/user-restricted-access-to-data),\nbut don't want to include it in the search results.\nAttribute names are case-sensitive.\n", "default": [], "x-categories": [ "Attributes" ] }, "disableTypoToleranceOnWords": { "type": "array", "items": { "type": "string" }, "description": "Creates a list of [words which require exact matches](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/typo-tolerance/in-depth/configuring-typo-tolerance/#turn-off-typo-tolerance-for-certain-words).\nThis also turns off [word splitting and concatenation](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/handling-natural-languages-nlp/in-depth/splitting-and-concatenation) for the specified words.\n", "default": [], "x-categories": [ "Typos" ] }, "attributesToTransliterate": { "description": "Attributes, for which you want to support [Japanese transliteration](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/handling-natural-languages-nlp/in-depth/language-specific-configurations/#japanese-transliteration-and-type-ahead).\n\nTransliteration supports searching in any of the Japanese writing systems.\nTo support transliteration, you must set the indexing language to Japanese.\nAttribute names are case-sensitive.\n", "type": "array", "items": { "type": "string" }, "x-categories": [ "Languages" ] }, "camelCaseAttributes": { "type": "array", "items": { "type": "string" }, "description": "Attributes for which to split [camel case](https://wikipedia.org/wiki/Camel_case) words.\nAttribute names are case-sensitive.\n", "default": [], "x-categories": [ "Languages" ] }, "decompoundedAttributes": { "type": "object", "description": "Searchable attributes to which Algolia should apply [word segmentation](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/handling-natural-languages-nlp/how-to/customize-segmentation) (decompounding).\nAttribute names are case-sensitive.\n\nCompound words are formed by combining two or more individual words,\nand are particularly prevalent in Germanic languages—for example, \"firefighter\".\nWith decompounding, the individual components are indexed separately.\n\nYou can specify different lists for different languages.\nDecompounding is supported for these languages:\nDutch (`nl`), German (`de`), Finnish (`fi`), Danish (`da`), Swedish (`sv`), and Norwegian (`no`).\nDecompounding doesn't work for words with [non-spacing mark Unicode characters](https://www.charactercodes.net/category/non-spacing_mark).\nFor example, `Gartenstühle` won't be decompounded if the `ü` consists of `u` (U+0075) and `◌̈` (U+0308).\n", "default": {}, "x-categories": [ "Languages" ] }, "indexLanguages": { "type": "array", "items": { "$ref": "#/$defs/supportedLanguage" }, "description": "Languages for language-specific processing steps, such as word detection and dictionary settings.\n\n**Always specify an indexing language.**\nIf you don't specify an indexing language, the search engine uses all [supported languages](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/handling-natural-languages-nlp/in-depth/supported-languages),\nor the languages you specified with the `ignorePlurals` or `removeStopWords` parameters.\nThis can lead to unexpected search results.\nFor more information, see [Language-specific configuration](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/handling-natural-languages-nlp/in-depth/language-specific-configurations).\n", "default": [], "x-categories": [ "Languages" ] }, "disablePrefixOnAttributes": { "type": "array", "items": { "type": "string" }, "description": "Searchable attributes for which you want to turn off [prefix matching](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/override-search-engine-defaults/#adjusting-prefix-search).\nAttribute names are case-sensitive.\n", "default": [], "x-categories": [ "Query strategy" ] }, "allowCompressionOfIntegerArray": { "type": "boolean", "description": "Whether arrays with exclusively non-negative integers should be compressed for better performance.\nIf true, the compressed arrays may be reordered.\n", "default": false, "x-categories": [ "Performance" ] }, "numericAttributesForFiltering": { "type": "array", "items": { "type": "string" }, "description": "Numeric attributes that can be used as [numerical filters](https://www.algolia.com/doc/guides/managing-results/rules/detecting-intent/how-to/applying-a-custom-filter-for-a-specific-query/#numerical-filters).\nAttribute names are case-sensitive.\n\nBy default, all numeric attributes are available as numerical filters.\nFor faster indexing, reduce the number of numeric attributes.\n\nTo turn off filtering for all numeric attributes, specify an attribute that doesn't exist in your index, such as `NO_NUMERIC_FILTERING`.\n\n**Modifier**\n\n- `equalOnly(\"ATTRIBUTE\")`.\n Support only filtering based on equality comparisons `=` and `!=`.\n", "default": [], "x-categories": [ "Performance" ] }, "separatorsToIndex": { "type": "string", "description": "Control which non-alphanumeric characters are indexed.\n\nBy default, Algolia ignores [non-alphanumeric characters](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/typo-tolerance/how-to/how-to-search-in-hyphenated-attributes/#handling-non-alphanumeric-characters) like hyphen (`-`), plus (`+`), and parentheses (`(`,`)`).\nTo include such characters, define them with `separatorsToIndex`.\n\nSeparators are all non-letter characters except spaces and currency characters, such as $€£¥.\n\nWith `separatorsToIndex`, Algolia treats separator characters as separate words.\nFor example, in a search for \"Disney+\", Algolia considers \"Disney\" and \"+\" as two separate words.\n", "default": "", "x-categories": [ "Typos" ] }, "searchableAttributes": { "type": "array", "items": { "type": "string" }, "description": "Attributes used for searching. Attribute names are case-sensitive.\n\nBy default, all attributes are searchable and the [Attribute](https://www.algolia.com/doc/guides/managing-results/relevance-overview/in-depth/ranking-criteria/#attribute) ranking criterion is turned off.\nWith a non-empty list, Algolia only returns results with matches in the selected attributes.\nIn addition, the Attribute ranking criterion is turned on: matches in attributes that are higher in the list of `searchableAttributes` rank first.\nTo make matches in two attributes rank equally, include them in a comma-separated string, such as `\"title,alternate_title\"`.\nAttributes with the same priority are always unordered.\n\nFor more information, see [Searchable attributes](https://www.algolia.com/doc/guides/sending-and-managing-data/prepare-your-data/how-to/setting-searchable-attributes).\n\n**Modifier**\n\n- `unordered(\"ATTRIBUTE\")`.\n Ignore the position of a match within the attribute.\n\nWithout a modifier, matches at the beginning of an attribute rank higher than matches at the end.\n", "default": [], "x-categories": [ "Attributes" ] }, "userData": { "$ref": "#/$defs/userData" }, "customNormalization": { "description": "Characters and their normalized replacements.\nThis overrides Algolia's default [normalization](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/handling-natural-languages-nlp/in-depth/normalization).\n", "type": "object", "additionalProperties": { "type": "object", "additionalProperties": { "type": "string" } }, "x-categories": [ "Languages" ] }, "attributeForDistinct": { "description": "Attribute that should be used to establish groups of results.\nAttribute names are case-sensitive.\n\nAll records with the same value for this attribute are considered a group.\nYou can combine `attributeForDistinct` with the `distinct` search parameter to control\nhow many items per group are included in the search results.\n\nIf you want to use the same attribute also for faceting, use the `afterDistinct` modifier of the `attributesForFaceting` setting.\nThis applies faceting _after_ deduplication, which will result in accurate facet counts.\n", "type": "string" }, "maxFacetHits": { "$ref": "#/$defs/maxFacetHits" }, "keepDiacriticsOnCharacters": { "type": "string", "description": "Characters for which diacritics should be preserved.\n\nBy default, Algolia removes diacritics from letters.\nFor example, `é` becomes `e`. If this causes issues in your search,\nyou can specify characters that should keep their diacritics.\n", "default": "", "x-categories": [ "Languages" ] }, "customRanking": { "type": "array", "items": { "type": "string" }, "description": "Attributes to use as [custom ranking](https://www.algolia.com/doc/guides/managing-results/must-do/custom-ranking).\nAttribute names are case-sensitive.\n\nThe custom ranking attributes decide which items are shown first if the other ranking criteria are equal.\n\nRecords with missing values for your selected custom ranking attributes are always sorted last.\nBoolean attributes are sorted based on their alphabetical order.\n\n**Modifiers**\n\n- `asc(\"ATTRIBUTE\")`.\n Sort the index by the values of an attribute, in ascending order.\n\n- `desc(\"ATTRIBUTE\")`.\n Sort the index by the values of an attribute, in descending order.\n\nIf you use two or more custom ranking attributes,\n[reduce the precision](https://www.algolia.com/doc/guides/managing-results/must-do/custom-ranking/how-to/controlling-custom-ranking-metrics-precision) of your first attributes,\nor the other attributes will never be applied.\n", "default": [], "x-categories": [ "Ranking" ] } } }, "beforeIndexPublishing": { "type": "object", "description": "Checks triggered after the crawl finishes but before the records are added to the Algolia index.", "properties": { "maxLostRecordsPercentage": { "type": "integer", "description": "Maximum difference in percent between the numbers of records between crawls.", "minimum": 1, "maximum": 100, "default": 10 }, "maxFailedUrls": { "type": "integer", "description": "Stops the crawler if a specified number of pages fail to crawl." } } }, "booleanString": { "type": "string", "enum": [ "true", "false" ] }, "browserRequest": { "title": "Browser-based", "type": "object", "description": "Information for using a web browser for authorization.\nThe browser loads a login page and enters the provided credentials.\n", "properties": { "url": { "type": "string", "description": "URL of your login page.\n\nThe crawler looks for an input matching the selector `input[type=text]` or `input[type=email]` for the username and `input[type=password]` for the password.\n" }, "username": { "type": "string", "description": "Username for signing in." }, "password": { "type": "string", "description": "Password for signing in." }, "waitTime": { "$ref": "#/$defs/waitTime" } }, "required": [ "url", "username", "password" ] }, "cache": { "type": "object", "description": "Whether the crawler should cache crawled pages.\n\nFor more information, see [Partial crawls with caching](https://www.algolia.com/doc/tools/crawler/getting-started/crawler-configuration/#partial-crawls-with-caching).\n", "properties": { "enabled": { "type": "boolean", "default": true, "description": "Whether the crawler cache is active." } } }, "configurationRecordExtractorType": { "type": "string", "enum": [ "function" ] }, "decompoundQuery": { "type": "boolean", "description": "Whether to split compound words in the query into their building blocks\nFor more information, see [Word segmentation](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/handling-natural-languages-nlp/in-depth/language-specific-configurations/#splitting-compound-words).\nWord segmentation is supported for these languages: German, Dutch, Finnish, Swedish, and Norwegian.\nDecompounding doesn't work for words with [non-spacing mark Unicode characters](https://www.charactercodes.net/category/non-spacing_mark).\nFor example, `Gartenstühle` won't be decompounded if the `ü` consists of `u` (U+0075) and `◌̈` (U+0308).\n", "default": true, "x-categories": [ "Languages" ] }, "disableExactOnAttributes": { "type": "array", "items": { "type": "string" }, "description": "Searchable attributes for which you want to [turn off the Exact ranking criterion](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/override-search-engine-defaults/in-depth/adjust-exact-settings/#turn-off-exact-for-some-attributes).\nAttribute names are case-sensitive\nThis can be useful for attributes with long values, where the likelihood of an exact match is high,\nsuch as product descriptions.\nTurning off the Exact ranking criterion for these attributes favors exact matching on other attributes.\nThis reduces the impact of individual attributes with a lot of content on ranking.\n", "default": [], "x-categories": [ "Query strategy" ] }, "disableTypoToleranceOnAttributes": { "type": "array", "items": { "type": "string" }, "description": "Attributes for which you want to turn off [typo tolerance](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/typo-tolerance).\nAttribute names are case-sensitive\nReturning only exact matches can help when\n- [Searching in hyphenated attributes](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/typo-tolerance/how-to/how-to-search-in-hyphenated-attributes).\n- Reducing the number of matches when you have too many.\n This can happen with attributes that are long blocks of text, such as product descriptions\nConsider alternatives such as `disableTypoToleranceOnWords` or adding synonyms if your attributes have intentional unusual spellings that might look like typos.\n", "default": [], "x-categories": [ "Typos" ] }, "distinct": { "description": "Determines how many records of a group are included in the search results.\n\nRecords with the same value for the `attributeForDistinct` attribute are considered a group.\nThe `distinct` setting controls how many members of the group are returned.\nThis is useful for [deduplication and grouping](https://www.algolia.com/doc/guides/managing-results/refine-results/grouping/#introducing-algolias-distinct-feature).\n\nThe `distinct` setting is ignored if `attributeForDistinct` is not set.\n", "oneOf": [ { "type": "boolean", "description": "Whether deduplication is turned on. If true, only one member of a group is shown in the search results." }, { "type": "integer", "description": "Number of members of a group of records to include in the search results.\n\n- Don't use `distinct > 1` for records that might be [promoted by rules](https://www.algolia.com/doc/guides/managing-results/rules/merchandising-and-promoting/how-to/promote-hits).\n The number of hits won't be correct and faceting won't work as expected.\n- With `distinct > 1`, the `hitsPerPage` parameter controls the number of returned groups.\n For example, with `hitsPerPage: 10` and `distinct: 2`, up to 20 records are returned.\n Likewise, the `nbHits` response attribute contains the number of returned groups.\n", "minimum": 0, "maximum": 4, "default": 0 } ], "x-categories": [ "Advanced" ] }, "enablePersonalization": { "type": "boolean", "description": "Whether to enable Personalization.", "default": false, "x-categories": [ "Personalization" ] }, "enableReRanking": { "type": "boolean", "description": "Whether this search will use [Dynamic Re-Ranking](https://www.algolia.com/doc/guides/algolia-ai/re-ranking)\nThis setting only has an effect if you activated Dynamic Re-Ranking for this index in the Algolia dashboard.\n", "default": true, "x-categories": [ "Filtering" ] }, "enableRules": { "type": "boolean", "description": "Whether to enable rules.", "default": true, "x-categories": [ "Rules" ] }, "exactOnSingleWordQuery": { "type": "string", "enum": [ "attribute", "none", "word" ], "description": "Determines how the [Exact ranking criterion](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/override-search-engine-defaults/in-depth/adjust-exact-settings/#turn-off-exact-for-some-attributes) is computed when the search query has only one word.\n\n- `attribute`.\n The Exact ranking criterion is 1 if the query word and attribute value are the same.\n For example, a search for \"road\" will match the value \"road\", but not \"road trip\".\n\n- `none`.\n The Exact ranking criterion is ignored on single-word searches.\n\n- `word`.\n The Exact ranking criterion is 1 if the query word is found in the attribute value.\n The query word must have at least 3 characters and must not be a stop word.\n Only exact matches will be highlighted,\n partial and prefix matches won't.\n", "default": "attribute", "x-categories": [ "Query strategy" ] }, "extraParameters": { "type": "object", "description": "Extra parameters for the authorization request.", "properties": { "resource": { "type": "string", "description": "App ID URI of the receiving web service.\n\nFor more information, see [Azure Active Directory](https://learn.microsoft.com/en-us/previous-versions/azure/active-directory/azuread-dev/v1-oauth2-client-creds-grant-flow#first-case-access-token-request-with-a-shared-secret).\n" } } }, "facetOrdering": { "description": "Order of facet names and facet values in your UI.", "type": "object", "additionalProperties": false, "properties": { "facets": { "$ref": "#/$defs/facets" }, "values": { "$ref": "#/$defs/values" } } }, "facets": { "description": "Order of facet names.", "type": "object", "additionalProperties": false, "properties": { "order": { "$ref": "#/$defs/order" } } }, "fetchRequest": { "title": "HTTP request", "type": "object", "description": "Information for making a HTTP request for authorization.", "properties": { "url": { "type": "string", "description": "URL with your login form." }, "requestOptions": { "$ref": "#/$defs/loginRequestOptions" } }, "required": [ "url" ] }, "fileTypes": { "type": "string", "description": "For more information, see [Extract data from non-HTML documents](https://www.algolia.com/doc/tools/crawler/extracting-data/non-html-documents).\n", "enum": [ "doc", "email", "html", "odp", "ods", "odt", "pdf", "ppt", "xls" ] }, "grantType": { "type": "string", "description": "OAuth 2.0 grant type.", "enum": [ "client_credentials" ] }, "headers": { "type": "object", "description": "Headers to add to all requests.", "properties": { "Accept-Language": { "type": "string", "description": "Preferred natural language and locale." }, "Authorization": { "type": "string", "description": "Basic authentication header." }, "Cookie": { "type": "string", "description": "Cookie. The header will be replaced by the cookie retrieved when logging in." } } }, "hide": { "description": "Hide facet values.", "type": "array", "items": { "type": "string" } }, "highlightPostTag": { "type": "string", "description": "HTML tag to insert after the highlighted parts in all highlighted results and snippets.", "default": "", "x-categories": [ "Highlighting and Snippeting" ] }, "highlightPreTag": { "type": "string", "description": "HTML tag to insert before the highlighted parts in all highlighted results and snippets.", "default": "", "x-categories": [ "Highlighting and Snippeting" ] }, "hitsPerPage": { "type": "integer", "description": "Number of hits per page.", "default": 20, "minimum": 1, "maximum": 1000, "x-categories": [ "Pagination" ] }, "hostnameAliases": { "type": "object", "description": "Key-value pairs to replace matching hostnames found in a sitemap,\non a page, in canonical links, or redirects.\n\n\nDuring a crawl, this action maps one hostname to another whenever the crawler encounters specific URLs.\nThis helps with links to staging environments (like `dev.example.com`) or external hosting services (such as YouTube).\n\n\nFor example, with this `hostnameAliases` mapping:\n\n {\n hostnameAliases: {\n 'dev.example.com': 'example.com'\n }\n }\n\n1. The crawler encounters `https://dev.example.com/solutions/voice-search/`.\n\n1. `hostnameAliases` transforms the URL to `https://example.com/solutions/voice-search/`.\n\n1. The crawler follows the transformed URL (not the original).\n\n\n**`hostnameAliases` only changes URLs, not page text. In the preceding example, if the extracted text contains the string `dev.example.com`, it remains unchanged.**\n\n\nThe crawler can discover URLs in places such as:\n\n\n- Crawled pages\n\n- Sitemaps\n\n- [Canonical URLs](https://www.algolia.com/doc/tools/crawler/getting-started/crawler-configuration/#canonical-urls-and-crawler-behavior)\n\n- Redirects. \n\n\nHowever, `hostnameAliases` doesn't transform URLs you explicitly set in the `startUrls` or `sitemaps` parameters,\nnor does it affect the `pathsToMatch` action or other configuration elements.\n", "additionalProperties": { "type": "string", "description": "Hostname that should be added in the records.", "x-additionalPropertiesName": "hostname" } }, "ignoreCanonicalTo": { "oneOf": [ { "type": "boolean", "description": "Determines if the crawler should extract records from a page with a [canonical URL](https://www.algolia.com/doc/tools/crawler/getting-started/crawler-configuration/#canonical-urls-and-crawler-behavior).\n\nIf `ignoreCanonicalTo` is set to:\n\n- `true` all canonical URLs are ignored.\n- One or more URL patterns, the crawler will ignore the canonical URL if it matches a pattern.\n" }, { "type": "array", "description": "Canonical URLs or URL patterns to ignore.\n", "items": { "type": "string", "description": "Pattern or URL.\n\nCanonical URLs are only ignored if they match this pattern.\n" } } ] }, "ignorePlurals": { "description": "Treat singular, plurals, and other forms of declensions as equivalent.\nOnly use this feature for the languages used in your index.\n", "oneOf": [ { "type": "array", "description": "ISO code for languages for which this feature should be active.\nThis overrides languages you set with `queryLanguages`.\n", "items": { "$ref": "#/$defs/supportedLanguage" } }, { "$ref": "#/$defs/booleanString" }, { "type": "boolean", "description": "If true, `ignorePlurals` is active for all languages included in `queryLanguages`, or for all supported languages, if `queryLanguges` is empty.\nIf false, singulars, plurals, and other declensions won't be considered equivalent.\n", "default": false } ], "x-categories": [ "Languages" ] }, "indexSettings": { "description": "Index settings.", "allOf": [ { "$ref": "#/$defs/baseIndexSettings" }, { "$ref": "#/$defs/indexSettingsAsSearchParams" } ] }, "indexSettingsAsSearchParams": { "type": "object", "additionalProperties": false, "properties": { "attributesToRetrieve": { "$ref": "#/$defs/attributesToRetrieve" }, "ranking": { "type": "array", "items": { "type": "string" }, "description": "Determines the order in which Algolia returns your results.\n\nBy default, each entry corresponds to a [ranking criteria](https://www.algolia.com/doc/guides/managing-results/relevance-overview/in-depth/ranking-criteria).\nThe tie-breaking algorithm sequentially applies each criterion in the order they're specified.\nIf you configure a replica index for [sorting by an attribute](https://www.algolia.com/doc/guides/managing-results/refine-results/sorting/how-to/sort-by-attribute),\nyou put the sorting attribute at the top of the list.\n\n**Modifiers**\n\n- `asc(\"ATTRIBUTE\")`.\n Sort the index by the values of an attribute, in ascending order.\n- `desc(\"ATTRIBUTE\")`.\n Sort the index by the values of an attribute, in descending order.\n\nBefore you modify the default setting,\ntest your changes in the dashboard,\nand by [A/B testing](https://www.algolia.com/doc/guides/ab-testing/what-is-ab-testing).\n", "default": [ "typo", "geo", "words", "filters", "proximity", "attribute", "exact", "custom" ], "x-categories": [ "Ranking" ] }, "relevancyStrictness": { "$ref": "#/$defs/relevancyStrictness" }, "attributesToHighlight": { "$ref": "#/$defs/attributesToHighlight" }, "attributesToSnippet": { "$ref": "#/$defs/attributesToSnippet" }, "highlightPreTag": { "$ref": "#/$defs/highlightPreTag" }, "highlightPostTag": { "$ref": "#/$defs/highlightPostTag" }, "snippetEllipsisText": { "$ref": "#/$defs/snippetEllipsisText" }, "restrictHighlightAndSnippetArrays": { "$ref": "#/$defs/restrictHighlightAndSnippetArrays" }, "hitsPerPage": { "$ref": "#/$defs/hitsPerPage" }, "minWordSizefor1Typo": { "$ref": "#/$defs/minWordSizefor1Typo" }, "minWordSizefor2Typos": { "$ref": "#/$defs/minWordSizefor2Typos" }, "typoTolerance": { "$ref": "#/$defs/typoTolerance" }, "allowTyposOnNumericTokens": { "$ref": "#/$defs/allowTyposOnNumericTokens" }, "disableTypoToleranceOnAttributes": { "$ref": "#/$defs/disableTypoToleranceOnAttributes" }, "ignorePlurals": { "$ref": "#/$defs/ignorePlurals" }, "removeStopWords": { "$ref": "#/$defs/removeStopWords" }, "queryLanguages": { "$ref": "#/$defs/queryLanguages" }, "decompoundQuery": { "$ref": "#/$defs/decompoundQuery" }, "enableRules": { "$ref": "#/$defs/enableRules" }, "enablePersonalization": { "$ref": "#/$defs/enablePersonalization" }, "queryType": { "$ref": "#/$defs/queryType" }, "removeWordsIfNoResults": { "$ref": "#/$defs/removeWordsIfNoResults" }, "mode": { "$ref": "#/$defs/mode" }, "semanticSearch": { "$ref": "#/$defs/semanticSearch" }, "advancedSyntax": { "$ref": "#/$defs/advancedSyntax" }, "optionalWords": { "$ref": "#/$defs/optionalWords" }, "disableExactOnAttributes": { "$ref": "#/$defs/disableExactOnAttributes" }, "exactOnSingleWordQuery": { "$ref": "#/$defs/exactOnSingleWordQuery" }, "alternativesAsExact": { "$ref": "#/$defs/IndexSettings_alternativesAsExact" }, "advancedSyntaxFeatures": { "$ref": "#/$defs/IndexSettings_advancedSyntaxFeatures" }, "distinct": { "$ref": "#/$defs/distinct" }, "replaceSynonymsInHighlight": { "$ref": "#/$defs/replaceSynonymsInHighlight" }, "minProximity": { "$ref": "#/$defs/minProximity" }, "responseFields": { "$ref": "#/$defs/responseFields" }, "maxValuesPerFacet": { "$ref": "#/$defs/maxValuesPerFacet" }, "sortFacetValuesBy": { "$ref": "#/$defs/sortFacetValuesBy" }, "attributeCriteriaComputedByMinProximity": { "$ref": "#/$defs/attributeCriteriaComputedByMinProximity" }, "renderingContent": { "$ref": "#/$defs/renderingContent" }, "enableReRanking": { "$ref": "#/$defs/enableReRanking" }, "reRankingApplyFilter": { "oneOf": [ { "$ref": "#/$defs/reRankingApplyFilter" }, { "type": "null" } ] } } }, "login": { "description": "Authorization method and credentials for crawling protected content.\n\nThe Crawler supports these authentication methods:\n\n- **Basic authentication**.\n The Crawler obtains a session cookie from the login page.\n- **OAuth 2.0 authentication** (`oauthRequest`).\n The Crawler uses OAuth 2.0 client credentials to obtain an access token for authentication.\n\n**Basic authentication**\n\nThe Crawler extracts the `Set-Cookie` response header from the login page, stores that cookie,\nand sends it in the `Cookie` header when crawling all pages defined in the configuration.\n\nThis cookie is retrieved only at the start of each full crawl.\nIf it expires, it isn't automatically renewed.\n\nThe Crawler can obtain the session cookie in one of two ways:\n\n- **HTTP request authentication** (`fetchRequest`).\n The Crawler sends a direct request with your credentials to the login endpoint, similar to a `curl` command.\n- **Browser-based authentication** (`browserRequest`).\n The Crawler emulates a web browser by loading the login page, entering the credentials,\n and submitting the login form as a real user would.\n\n**OAuth 2.0**\n\nThe crawler supports [OAuth 2.0 client credentials grant flow](https://datatracker.ietf.org/doc/html/rfc6749#section-4.4):\n\n1. It performs an access token request with the provided credentials\n1. Stores the fetched token in an `Authorization` header\n1. Sends the token when crawling site pages.\n\nThis token is only fetched at the beginning of each complete crawl.\nIf it expires, it isn't automatically renewed.\n\nClient authentication passes the credentials (`client_id` and `client_secret`) [in the request body](https://datatracker.ietf.org/doc/html/rfc6749#section-2.3.1).\nThe [Azure AD v1.0](https://learn.microsoft.com/en-us/previous-versions/azure/active-directory/azuread-dev/v1-oauth2-client-creds-grant-flow) provider is supported.\n", "oneOf": [ { "$ref": "#/$defs/fetchRequest" }, { "$ref": "#/$defs/browserRequest" }, { "$ref": "#/$defs/oauthRequest" } ] }, "loginRequestOptions": { "type": "object", "description": "Options for the HTTP request for logging in.", "properties": { "method": { "type": "string", "description": "HTTP method for sending the request.", "default": "GET" }, "headers": { "$ref": "#/$defs/headers" }, "body": { "type": "string", "description": "Form content." }, "timeout": { "type": "integer", "description": "Timeout for the request." } } }, "maxFacetHits": { "type": "integer", "description": "Maximum number of facet values to return when [searching for facet values](https://www.algolia.com/doc/guides/managing-results/refine-results/faceting/#search-for-facet-values).", "maximum": 100, "default": 10, "x-categories": [ "Advanced" ] }, "maxValuesPerFacet": { "type": "integer", "description": "Maximum number of facet values to return for each facet.", "default": 100, "maximum": 1000, "x-categories": [ "Faceting" ] }, "minProximity": { "type": "integer", "minimum": 1, "maximum": 7, "description": "Minimum proximity score for two matching words\nThis adjusts the [Proximity ranking criterion](https://www.algolia.com/doc/guides/managing-results/relevance-overview/in-depth/ranking-criteria/#proximity)\nby equally scoring matches that are farther apart\nFor example, if `minProximity` is 2, neighboring matches and matches with one word between them would have the same score.\n", "default": 1, "x-categories": [ "Advanced" ] }, "minWordSizefor1Typo": { "type": "integer", "description": "Minimum number of characters a word in the search query must contain to accept matches with [one typo](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/typo-tolerance/in-depth/configuring-typo-tolerance/#configuring-word-length-for-typos).", "default": 4, "x-categories": [ "Typos" ] }, "minWordSizefor2Typos": { "type": "integer", "description": "Minimum number of characters a word in the search query must contain to accept matches with [two typos](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/typo-tolerance/in-depth/configuring-typo-tolerance/#configuring-word-length-for-typos).", "default": 8, "x-categories": [ "Typos" ] }, "mode": { "type": "string", "enum": [ "neuralSearch", "keywordSearch" ], "description": "Search mode the index will use to query for results.\n\nThis setting only applies to indices, for which Algolia enabled NeuralSearch for you.\n", "default": "keywordSearch", "x-categories": [ "Query strategy" ] }, "oauthRequest": { "title": "OAuth 2.0", "type": "object", "description": "Authorization information for using the [OAuth 2.0 client credentials](https://datatracker.ietf.org/doc/html/rfc6749#section-4.4) authorization grant.\n\nOAuth authorization is supported for [Azure Active Directory version 1](https://learn.microsoft.com/en-us/previous-versions/azure/active-directory/azuread-dev/v1-oauth2-client-creds-grant-flow) as provider.\n", "properties": { "accessTokenRequest": { "$ref": "#/$defs/accessTokenRequest" } }, "required": [ "accessTokenRequest" ] }, "optionalWords": { "description": "Words that should be considered optional when found in the query.\n\nBy default, records must match all words in the search query to be included in the search results.\nAdding optional words can increase the number of search results by running an additional search query that doesn't include the optional words.\nFor example, if the search query is \"action video\" and \"video\" is optional,\nthe search engine runs two queries: one for \"action video\" and one for \"action\".\nRecords that match all words are ranked higher.\n\nFor a search query with 4 or more words **and** all its words are optional,\nthe number of matched words required for a record to be included in the search results increases for every 1,000 records:\n\n- If `optionalWords` has fewer than 10 words, the required number of matched words increases by 1:\n results 1 to 1,000 require 1 matched word; results 1,001 to 2,000 need 2 matched words.\n- If `optionalWords` has 10 or more words, the required number of matched words increases by the number of optional words divided by 5 (rounded down).\n Example: with 18 optional words, results 1 to 1,000 require 1 matched word; results 1,001 to 2,000 need 4 matched words.\n\nFor more information, see [Optional words](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/empty-or-insufficient-results/#creating-a-list-of-optional-words).\n", "oneOf": [ { "type": "string" }, { "type": "null" }, { "$ref": "#/$defs/optionalWordsArray" } ] }, "optionalWordsArray": { "type": "array", "items": { "type": "string" }, "description": "List of [optional words](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/empty-or-insufficient-results/#creating-a-list-of-optional-words).", "default": [], "x-categories": [ "Query strategy" ] }, "order": { "description": "Explicit order of facets or facet values.\n\nThis setting lets you always show specific facets or facet values at the top of the list.\n", "type": "array", "items": { "type": "string" } }, "pathAliases": { "type": "object", "description": "Key-value pairs to replace matching paths with new values.\n\nIt doesn't replace:\n\n- URLs in the `startUrls`, `sitemaps`, `pathsToMatch`, and other settings\n- Paths found in extracted text\n\nThe crawl continues from the _transformed_ URLs.\n\n\nFor example, if you create a mapping for `{ \"dev.example.com\": { '/foo': '/bar' } }` and the crawler encounters `https://dev.example.com/foo/hello/`,\nit’s transformed to `https://dev.example.com/bar/hello/`.\n\n\n> Compare with the `hostnameAliases` action.\n", "additionalProperties": { "type": "object", "description": "Hostname for which matching paths should be replaced.", "x-additionalPropertiesName": "hostname", "additionalProperties": { "type": "string", "description": "Key-value pair of a path that should be replaced.", "x-additionalPropertiesName": "path" } } }, "queryLanguages": { "type": "array", "items": { "$ref": "#/$defs/supportedLanguage" }, "description": "Languages for language-specific query processing steps such as plurals, stop-word removal, and word-detection dictionaries.\nThis setting sets a default list of languages used by the `removeStopWords` and `ignorePlurals` settings.\nThis setting also sets a dictionary for word detection in the logogram-based [CJK](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/handling-natural-languages-nlp/in-depth/normalization/#normalization-for-logogram-based-languages-cjk) languages.\nTo support this, place the CJK language **first**.\n**Always specify a query language.**\nIf you don't specify an indexing language, the search engine uses all [supported languages](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/handling-natural-languages-nlp/in-depth/supported-languages),\nor the languages you specified with the `ignorePlurals` or `removeStopWords` parameters.\nThis can lead to unexpected search results.\nFor more information, see [Language-specific configuration](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/handling-natural-languages-nlp/in-depth/language-specific-configurations).\n", "default": [], "x-categories": [ "Languages" ] }, "queryType": { "type": "string", "enum": [ "prefixLast", "prefixAll", "prefixNone" ], "description": "Determines if and how query words are interpreted as prefixes.\n\nBy default, only the last query word is treated as a prefix (`prefixLast`).\nTo turn off prefix search, use `prefixNone`.\nAvoid `prefixAll`, which treats all query words as prefixes.\nThis might lead to counterintuitive results and makes your search slower.\n\nFor more information, see [Prefix searching](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/override-search-engine-defaults/in-depth/prefix-searching).\n", "default": "prefixLast", "x-categories": [ "Query strategy" ] }, "reRankingApplyFilter": { "description": "Restrict [Dynamic Re-Ranking](https://www.algolia.com/doc/guides/algolia-ai/re-ranking) to records that match these filters.\n", "oneOf": [ { "type": "array", "items": { "$ref": "#/$defs/reRankingApplyFilter" } }, { "type": "string", "x-categories": [ "Filtering" ] } ] }, "redirectURL": { "description": "The redirect rule container.", "type": "object", "additionalProperties": false, "properties": { "url": { "type": "string" } } }, "relevancyStrictness": { "type": "integer", "description": "Relevancy threshold below which less relevant results aren't included in the results\nYou can only set `relevancyStrictness` on [virtual replica indices](https://www.algolia.com/doc/guides/managing-results/refine-results/sorting/in-depth/replicas/#what-are-virtual-replicas).\nUse this setting to strike a balance between the relevance and number of returned results.\n", "default": 100, "x-categories": [ "Ranking" ] }, "removeStopWords": { "description": "Removes stop words from the search query.\n\nStop words are common words like articles, conjunctions, prepositions, or pronouns that have little or no meaning on their own.\nIn English, \"the\", \"a\", or \"and\" are stop words.\n\nOnly use this feature for the languages used in your index.\n", "oneOf": [ { "type": "array", "description": "ISO code for languages for which stop words should be removed. This overrides languages you set in `queryLanguges`.", "items": { "$ref": "#/$defs/supportedLanguage" } }, { "type": "boolean", "default": false, "description": "If true, stop words are removed for all languages you included in `queryLanguages`, or for all supported languages, if `queryLanguages` is empty.\nIf false, stop words are not removed.\n" } ], "x-categories": [ "Languages" ] }, "removeWordsIfNoResults": { "type": "string", "enum": [ "none", "lastWords", "firstWords", "allOptional" ], "description": "Strategy for removing words from the query when it doesn't return any results.\nThis helps to avoid returning empty search results.\n\n- `none`.\n No words are removed when a query doesn't return results.\n\n- `lastWords`.\n Treat the last (then second to last, then third to last) word as optional,\n until there are results or at most 5 words have been removed.\n\n- `firstWords`.\n Treat the first (then second, then third) word as optional,\n until there are results or at most 5 words have been removed.\n\n- `allOptional`.\n Treat all words as optional.\n\nFor more information, see [Remove words to improve results](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/empty-or-insufficient-results/in-depth/why-use-remove-words-if-no-results).\n", "default": "none", "x-categories": [ "Query strategy" ] }, "renderJavaScript": { "description": "If `true`, use a Chrome headless browser to crawl pages.\n\nBecause crawling JavaScript-based web pages is slower than crawling regular HTML pages, you can apply this setting to a specific list of pages. \nUse [micromatch](https://github.com/micromatch/micromatch) to define URL patterns, including negations and wildcards.\n", "oneOf": [ { "type": "boolean", "description": "Whether to render all pages." }, { "type": "array", "description": "URLs or URL patterns to render.", "items": { "type": "string", "description": "URL or URL pattern to render." } }, { "title": "headlessBrowserConfig", "type": "object", "description": "Configuration for rendering HTML.", "properties": { "enabled": { "type": "boolean", "description": "Whether to enable JavaScript rendering." }, "patterns": { "type": "array", "description": "URLs or URL patterns to render.", "items": { "type": "string" } }, "adBlock": { "type": "boolean", "default": false, "description": "Whether to use the Crawler's ad blocker.\nIt blocks most ads and tracking scripts but can break some sites.\n" }, "waitTime": { "$ref": "#/$defs/waitTime" } }, "required": [ "enabled", "patterns" ] } ] }, "renderingContent": { "description": "Extra data that can be used in the search UI.\n\nYou can use this to control aspects of your search UI, such as the order of facet names and values\nwithout changing your frontend code.\n", "type": "object", "additionalProperties": false, "properties": { "facetOrdering": { "$ref": "#/$defs/facetOrdering" }, "redirect": { "$ref": "#/$defs/redirectURL" }, "widgets": { "$ref": "#/$defs/widgets" } }, "x-categories": [ "Advanced" ] }, "replaceSynonymsInHighlight": { "type": "boolean", "description": "Whether to replace a highlighted word with the matched synonym\nBy default, the original words are highlighted even if a synonym matches.\nFor example, with `home` as a synonym for `house` and a search for `home`,\nrecords matching either \"home\" or \"house\" are included in the search results,\nand either \"home\" or \"house\" are highlighted\nWith `replaceSynonymsInHighlight` set to `true`, a search for `home` still matches the same records,\nbut all occurrences of \"house\" are replaced by \"home\" in the highlighted response.\n", "default": false, "x-categories": [ "Highlighting and Snippeting" ] }, "requestOptions": { "type": "object", "description": "Lets you add options to HTTP requests made by the crawler.", "properties": { "proxy": { "type": "string", "description": "Proxy for all crawler requests." }, "timeout": { "type": "integer", "default": 30000, "description": "Timeout in milliseconds for the crawl." }, "retries": { "type": "integer", "default": 3, "description": "Maximum number of retries to crawl one URL." }, "headers": { "$ref": "#/$defs/headers" } } }, "responseFields": { "type": "array", "items": { "type": "string" }, "description": "Properties to include in the API response of search and browse requests\nBy default, all response properties are included.\nTo reduce the response size, you can select which properties should be included\nAn empty list may lead to an empty API response (except properties you can't exclude)\nYou can't exclude these properties:\n`message`, `warning`, `cursor`, `abTestVariantID`,\nor any property added by setting `getRankingInfo` to true\nYour search depends on the `hits` field. If you omit this field, searches won't return any results.\nYour UI might also depend on other properties, for example, for pagination.\nBefore restricting the response size, check the impact on your search experience.\n", "default": [ "*" ], "x-categories": [ "Advanced" ] }, "restrictHighlightAndSnippetArrays": { "type": "boolean", "description": "Whether to restrict highlighting and snippeting to items that at least partially matched the search query.\nBy default, all items are highlighted and snippeted.\n", "default": false, "x-categories": [ "Highlighting and Snippeting" ] }, "safetyChecks": { "type": "object", "description": "Checks to ensure the crawl was successful.\n\nFor more information, see the [Safety checks](https://www.algolia.com/doc/tools/crawler/getting-started/crawler-configuration/#safety-checks) documentation.\n", "properties": { "beforeIndexPublishing": { "$ref": "#/$defs/beforeIndexPublishing" } } }, "schedule": { "type": "string", "description": "Schedule for running the crawl.\n\nInstead of manually starting a crawl each time, you can set up a schedule for automatic crawls.\n[Use the visual UI](https://www.algolia.com/doc/tools/crawler/getting-started/crawler-configuration-visual) or add the `schedule` parameter to [your configuration](https://www.algolia.com/doc/tools/crawler/getting-started/crawler-configuration).\n\n`schedule` uses [Later.js syntax](https://bunkat.github.io/later) to specify when to crawl your site.\nHere are some key things to keep in mind when using `Later.js` syntax with the Crawler:\n\n- The interval between two scheduled crawls must be at least 24 hours.\n- To crawl daily, use \"every 1 day\" instead of \"everyday\" or \"every day\".\n- If you don't specify a time, the crawl can happen any time during the scheduled day.\n- Specify times for the UTC (GMT+0) timezone\n- Include minutes when specifying a time. For example, \"at 3:00 pm\" instead of \"at 3pm\".\n- Use \"at 12:00 am\" to specify midnight, not \"at 00:00 am\".\n" }, "semanticSearch": { "type": "object", "description": "Settings for the semantic search part of NeuralSearch.\nOnly used when `mode` is `neuralSearch`.\n", "properties": { "eventSources": { "oneOf": [ { "type": "array", "description": "Indices from which to collect click and conversion events.\n\nIf null, the current index and all its replicas are used.\n", "items": { "type": "string" } }, { "type": "null" } ] } } }, "snippetEllipsisText": { "type": "string", "description": "String used as an ellipsis indicator when a snippet is truncated.", "default": "…", "x-categories": [ "Highlighting and Snippeting" ] }, "sortFacetValuesBy": { "type": "string", "description": "Order in which to retrieve facet values\n- `count`.\n Facet values are retrieved by decreasing count.\n The count is the number of matching records containing this facet value\n- `alpha`.\n Retrieve facet values alphabetically\nThis setting doesn't influence how facet values are displayed in your UI (see `renderingContent`).\nFor more information, see [facet value display](https://www.algolia.com/doc/guides/building-search-ui/ui-and-ux-patterns/facet-display/js).\n", "default": "count", "x-categories": [ "Faceting" ] }, "sortRemainingBy": { "description": "Order of facet values that aren't explicitly positioned with the `order` setting.\n\n- `count`.\n Order remaining facet values by decreasing count.\n The count is the number of matching records containing this facet value.\n\n- `alpha`.\n Sort facet values alphabetically.\n\n- `hidden`.\n Don't show facet values that aren't explicitly positioned.\n", "type": "string", "enum": [ "count", "alpha", "hidden" ] }, "supportedLanguage": { "type": "string", "description": "ISO code for a supported language.", "enum": [ "af", "ar", "az", "bg", "bn", "ca", "cs", "cy", "da", "de", "el", "en", "eo", "es", "et", "eu", "fa", "fi", "fo", "fr", "ga", "gl", "he", "hi", "hu", "hy", "id", "is", "it", "ja", "ka", "kk", "ko", "ku", "ky", "lt", "lv", "mi", "mn", "mr", "ms", "mt", "nb", "nl", "no", "ns", "pl", "ps", "pt", "pt-br", "qu", "ro", "ru", "sk", "sq", "sv", "sw", "ta", "te", "th", "tl", "tn", "tr", "tt", "uk", "ur", "uz", "zh" ] }, "typoTolerance": { "description": "Whether [typo tolerance](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/typo-tolerance) is enabled and how it is applied.\n\nIf typo tolerance is true, `min`, or `strict`, [word splitting and concatenation](https://www.algolia.com/doc/guides/managing-results/optimize-search-results/handling-natural-languages-nlp/in-depth/splitting-and-concatenation) are also active.\n", "oneOf": [ { "type": "boolean", "default": true, "description": "Whether typo tolerance is active. If true, matches with typos are included in the search results and rank after exact matches." }, { "$ref": "#/$defs/typoToleranceEnum" } ], "x-categories": [ "Typos" ] }, "typoToleranceEnum": { "type": "string", "title": "typo tolerance", "description": "- `min`. Return matches with the lowest number of typos.\n For example, if you have matches without typos, only include those.\n But if there are no matches without typos (with 1 typo), include matches with 1 typo (2 typos).\n- `strict`. Return matches with the two lowest numbers of typos.\n With `strict`, the Typo ranking criterion is applied first in the `ranking` setting.\n", "enum": [ "min", "strict", "true", "false" ] }, "urlPattern": { "type": "string", "description": "Use [micromatch](https://github.com/micromatch/micromatch) for negation, wildcards, and more.\n" }, "userData": { "description": "An object with custom data.\n\nYou can store up to 32kB as custom data.\n", "default": {}, "x-categories": [ "Advanced" ] }, "value": { "type": "object", "additionalProperties": false, "properties": { "order": { "$ref": "#/$defs/order" }, "sortRemainingBy": { "$ref": "#/$defs/sortRemainingBy" }, "hide": { "$ref": "#/$defs/hide" } } }, "values": { "description": "Order of facet values. One object for each facet.", "type": "object", "additionalProperties": { "x-additionalPropertiesName": "facet", "$ref": "#/$defs/value" } }, "waitTime": { "type": "object", "description": "Timeout for the HTTP request.", "properties": { "min": { "type": "integer", "default": 0, "description": "Minimum waiting time in milliseconds." }, "max": { "type": "integer", "default": 20000, "description": "Maximum waiting time in milliseconds." } } }, "widgets": { "description": "Widgets returned from any rules that are applied to the current search.", "type": "object", "additionalProperties": false, "properties": { "banners": { "$ref": "#/$defs/banners" } } } } }