This commit is contained in:
aecsocket
2026-07-31 22:12:49 +02:00
committed by Michael H.
parent 4a2388f311
commit d658484f6b
@@ -233,7 +233,7 @@ impl ElasticsearchClient {
.request( .request(
Method::POST, Method::POST,
&format!( &format!(
"/{index}/_delete_by_query?conflicts=proceed&refresh=true" "/{index}/_delete_by_query?conflicts=proceed&refresh=false"
), ),
) )
.json(&json!({"query": query})) .json(&json!({"query": query}))
@@ -263,6 +263,21 @@ impl ElasticsearchClient {
response_json(response, "refresh Elasticsearch index").await?; response_json(response, "refresh Elasticsearch index").await?;
Ok(()) Ok(())
} }
async fn force_merge(&self, index: &str) -> Result<()> {
let response = self
.request(
Method::POST,
&format!(
"/{index}/_forcemerge?max_num_segments=1&flush=true"
),
)
.send()
.await
.wrap_err("failed to force-merge Elasticsearch index")?;
response_json(response, "force-merge Elasticsearch index").await?;
Ok(())
}
} }
async fn response_json( async fn response_json(
@@ -301,7 +316,51 @@ impl Elasticsearch {
"settings": { "settings": {
"number_of_shards": 3, "number_of_shards": 3,
"number_of_replicas": 1, "number_of_replicas": 1,
"index.mapping.total_fields.limit": 5000 "refresh_interval": "30s",
"index.mapping.total_fields.limit": 5000,
"analysis": {
"char_filter": {
"hyphen_separator": {
"type": "pattern_replace",
"pattern": "-",
"replacement": " "
},
"strip_symbols": {
"type": "pattern_replace",
"pattern": r"[^\p{L}\p{N}\s]",
"replacement": ""
}
},
"filter": {
"typesense_stemmer": {
"type": "stemmer",
"language": "light_english"
}
},
"analyzer": {
"typesense_text": {
"type": "custom",
"char_filter": ["strip_symbols"],
"tokenizer": "whitespace",
"filter": ["lowercase"]
},
"typesense_hyphen_text": {
"type": "custom",
"char_filter": [
"hyphen_separator",
"strip_symbols"
],
"tokenizer": "whitespace",
"filter": ["lowercase"]
},
"typesense_stemmed_text": {
"type": "custom",
"char_filter": ["strip_symbols"],
"tokenizer": "whitespace",
"filter": ["lowercase", "typesense_stemmer"]
}
}
}
}, },
"mappings": { "mappings": {
"dynamic_templates": [ "dynamic_templates": [
@@ -318,7 +377,8 @@ impl Elasticsearch {
"properties": { "properties": {
"document_type": { "document_type": {
"type": "join", "type": "join",
"relations": {"project": "version"} "relations": {"project": "version"},
"eager_global_ordinals": true
}, },
"version_id": {"type": "keyword"}, "version_id": {"type": "keyword"},
"project_id": {"type": "keyword"}, "project_id": {"type": "keyword"},
@@ -326,26 +386,72 @@ impl Elasticsearch {
"all_project_types": {"type": "keyword"}, "all_project_types": {"type": "keyword"},
"slug": { "slug": {
"type": "text", "type": "text",
"analyzer": "typesense_text",
"index_options": "docs",
"norms": false,
"index_prefixes": {
"min_chars": 1,
"max_chars": 10
},
"fields": { "fields": {
"keyword": {"type": "keyword", "ignore_above": 8191} "keyword": {"type": "keyword", "ignore_above": 8191}
} }
}, },
"author": { "author": {
"type": "text", "type": "text",
"analyzer": "typesense_hyphen_text",
"index_options": "docs",
"norms": false,
"index_prefixes": {
"min_chars": 1,
"max_chars": 10
},
"fields": { "fields": {
"keyword": {"type": "keyword", "ignore_above": 8191} "keyword": {"type": "keyword", "ignore_above": 8191}
} }
}, },
"indexed_author": {"type": "text"}, "indexed_author": {
"type": "text",
"analyzer": "typesense_text",
"index_options": "docs",
"norms": false,
"index_prefixes": {
"min_chars": 1,
"max_chars": 10
}
},
"name": { "name": {
"type": "text", "type": "text",
"analyzer": "typesense_hyphen_text",
"index_options": "docs",
"norms": false,
"index_prefixes": {
"min_chars": 1,
"max_chars": 10
},
"fields": { "fields": {
"keyword": {"type": "keyword", "ignore_above": 8191} "keyword": {"type": "keyword", "ignore_above": 8191}
} }
}, },
"indexed_name": {"type": "text"}, "indexed_name": {
"type": "text",
"analyzer": "typesense_stemmed_text",
"index_options": "docs",
"norms": false,
"index_prefixes": {
"min_chars": 1,
"max_chars": 10
}
},
"summary": { "summary": {
"type": "text", "type": "text",
"analyzer": "typesense_text",
"index_options": "docs",
"norms": false,
"index_prefixes": {
"min_chars": 1,
"max_chars": 10
},
"fields": { "fields": {
"keyword": {"type": "keyword", "ignore_above": 8191} "keyword": {"type": "keyword", "ignore_above": 8191}
} }
@@ -395,38 +501,109 @@ impl Elasticsearch {
return json!({"match_all": {}}); return json!({"match_all": {}});
} }
let tokens = query.split_whitespace().collect_vec();
if tokens.is_empty() {
return json!({"match_all": {}});
}
let fields = [ let fields = [
"name^15", ("name", 15),
"indexed_name^15", ("indexed_name", 15),
"slug^10", ("slug", 10),
"author^3", ("author", 3),
"indexed_author^3", ("indexed_author", 3),
"summary", ("summary", 1),
]; ];
json!({ let token_queries =
"bool": { |token: &str, prefix: bool, exact_boost: bool| {
"should": [ let mut queries = Vec::with_capacity(fields.len() * 3);
{ for (field, weight) in fields {
"multi_match": { if exact_boost && field != "indexed_name" {
"query": query, queries.push(json!({
"fields": fields, "constant_score": {
"type": "best_fields", "filter": {
"operator": "and", "match": {
"fuzziness": "AUTO", (field): {
"prefix_length": 1 "query": token,
"fuzziness": 0
}
} }
}, },
{ "boost": weight + 1
"multi_match": { }
"query": query, }));
"fields": fields, }
"type": "phrase_prefix" if prefix {
queries.push(json!({
"constant_score": {
"filter": {
"match_bool_prefix": {
(field): {"query": token}
}
},
"boost": weight
}
}));
}
queries.push(json!({
"constant_score": {
"filter": {
"match": {
(field): {
"query": token,
"fuzziness": "AUTO:4,7",
"prefix_length": 1,
"max_expansions": 2
} }
} }
], },
"minimum_should_match": 1 "boost": weight
}
}));
}
queries
};
let queries_by_token = tokens
.iter()
.enumerate()
.map(|(index, token)| {
// Typesense caps candidates globally, while Elasticsearch
// expands them per field and shard. These bounds keep broad
// prefixes and stem-only matches out of the top relevance
// bucket while preserving short autocomplete queries.
let is_last = index == tokens.len() - 1;
let prefix =
is_last && (tokens.len() == 1 || token.chars().count() < 6);
let exact_boost =
tokens.len() > 1 || token.chars().count() >= 6;
token_queries(token, prefix, exact_boost)
})
.collect_vec();
let scoring_query = json!({
"dis_max": {
"queries": queries_by_token.iter().flatten().collect_vec(),
"tie_breaker": 0
}
});
if tokens.len() == 1 {
scoring_query
} else {
json!({
"bool": {
"must": [scoring_query],
"filter": queries_by_token
.into_iter()
.map(|queries| {
json!({
"dis_max": {
"queries": queries,
"tie_breaker": 0
} }
}) })
})
.collect_vec()
}
})
}
} }
fn sort(index: SearchIndex) -> Vec<Value> { fn sort(index: SearchIndex) -> Vec<Value> {
@@ -729,6 +906,8 @@ impl SearchBackend for Elasticsearch {
} }
self.client.refresh(&next).await?; self.client.refresh(&next).await?;
info!("force-merging Elasticsearch shadow index");
self.client.force_merge(&next).await?;
info!("swapping Elasticsearch index alias"); info!("swapping Elasticsearch index alias");
self.client self.client
.swap_alias(&alias, current.as_deref(), &next) .swap_alias(&alias, current.as_deref(), &next)
@@ -757,13 +936,6 @@ impl SearchBackend for Elasticsearch {
.removed_versions .removed_versions
.iter() .iter()
.map(ToString::to_string) .map(ToString::to_string)
.chain(
update
.versions
.iter()
.map(|document| document.version_id.clone()),
)
.unique()
.collect::<Vec<_>>(); .collect::<Vec<_>>();
if !version_ids.is_empty() { if !version_ids.is_empty() {
self.delete_ids("version_id", &version_ids).await?; self.delete_ids("version_id", &version_ids).await?;