fix: search indexing performance and batching (#6521)

* Add consume batching delay

* maybe fix

* max batch size

* more logging

* parallelize remove tasks

* delete/upsert project/version messages

* prepare

* more logging

* log number of docs

* try more targeted, homogenous version change ops

* ensure only necessary fields are serialized into typesense

* disable index background task

* wip: script changes

* don't facet by project id

* fix

* fix clippy

* batch by document and dedup loaders

* fix test

* wip: projects/versions collections

* clean up SearchBackend interface

* cleanup pass

* cleanup pass 2

* standardise fn names

* cleanup pass

* fix compile

* factor out filter rewriting

* query perf

* put categories into the project doc

* wip: search filter AST

* doc comment

* more AST normalization

* (temp) convert search request errors into internal errors

* implement unary NOT

* tombi fmt

* fix tests

* revert request error

* try increase stack size
This commit is contained in:
aecsocket
2026-07-26 14:55:00 +00:00
committed by GitHub
parent 54cc1898fa
commit 158de019a6
32 changed files with 2787 additions and 720 deletions
+49 -11
View File
@@ -16,6 +16,7 @@ use utoipa::ToSchema;
use xredis::RedisPool;
pub mod backend;
pub mod filter;
pub mod incremental;
pub mod indexing;
@@ -105,24 +106,17 @@ pub trait SearchBackend: Send + Sync {
info: &SearchRequest,
) -> Result<SearchResults, ApiError>;
async fn index_projects(
async fn rebuild_index(
&self,
ro_pool: PgPool,
redis: RedisPool,
) -> eyre::Result<()>;
async fn index_documents(
async fn apply_update(
&self,
documents: &[UploadSearchProject],
update: SearchIndexUpdate<'_>,
) -> eyre::Result<()>;
async fn remove_project_documents(
&self,
ids: &[ProjectId],
) -> eyre::Result<()>;
async fn remove_documents(&self, ids: &[VersionId]) -> eyre::Result<()>;
async fn tasks(&self) -> eyre::Result<Value>;
async fn tasks_cancel(
@@ -235,8 +229,12 @@ impl FromStr for SearchBackendKind {
}
}
/// Nullable fields in Typesense-bound documents should use
/// `skip_serializing_if = "Option::is_none"` so they are omitted instead of
/// serialized as `null`.
#[derive(Serialize, Deserialize, Debug, Clone)]
pub struct UploadSearchProject {
/// ID of the most recently published version.
pub version_id: String,
pub project_id: String,
//
@@ -246,20 +244,25 @@ pub struct UploadSearchProject {
pub slug: Option<String>,
pub author: String,
pub author_id: String,
#[serde(skip_serializing_if = "Option::is_none")]
pub organization: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub organization_id: Option<String>,
pub indexed_author: String,
pub name: String,
pub indexed_name: String,
pub summary: String,
pub categories: Vec<String>,
pub project_categories: Vec<String>,
pub display_categories: Vec<String>,
pub follows: i32,
pub downloads: i32,
pub log_downloads: f64,
#[serde(skip_serializing_if = "Option::is_none")]
pub icon_url: Option<String>,
pub license: String,
pub gallery: Vec<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub featured_gallery: Option<String>,
/// RFC 3339 formatted creation date of the project
pub date_created: DateTime<Utc>,
@@ -269,9 +272,10 @@ pub struct UploadSearchProject {
pub date_modified: DateTime<Utc>,
/// Unix timestamp of the last major modification
pub modified_timestamp: i64,
/// Unix timestamp of the publication date of the version
/// Unix timestamp of the most recently published version.
pub version_published_timestamp: i64,
pub open_source: bool,
#[serde(skip_serializing_if = "Option::is_none")]
pub color: Option<u32>,
#[serde(default)]
pub dependency_project_ids: Vec<String>,
@@ -290,12 +294,45 @@ pub struct UploadSearchProject {
pub loader_fields: HashMap<String, Vec<serde_json::Value>>,
}
#[derive(Serialize, Deserialize, Debug, Clone)]
pub struct UploadSearchVersion {
pub version_id: String,
pub project_id: String,
pub categories: Vec<String>,
pub project_types: Vec<String>,
pub version_published_timestamp: i64,
#[serde(flatten)]
pub loader_fields: HashMap<String, Vec<serde_json::Value>>,
}
#[derive(Debug, Default)]
pub struct SearchDocumentBatch {
pub projects: Vec<UploadSearchProject>,
pub versions: Vec<UploadSearchVersion>,
}
/// A logical search index mutation. Removals are applied before replacements,
/// so a document may be present in both a removed and replacement field.
#[derive(Debug, Clone, Copy, Default)]
pub struct SearchIndexUpdate<'a> {
pub projects: &'a [UploadSearchProject],
pub versions: &'a [UploadSearchVersion],
/// Projects and all of their version documents to remove.
pub removed_projects: &'a [ProjectId],
pub removed_versions: &'a [VersionId],
}
/// Nullable fields in Typesense-bound documents should use
/// `skip_serializing_if = "Option::is_none"` so they are omitted instead of
/// serialized as `null`.
#[derive(Serialize, Deserialize, Debug, Clone, ToSchema)]
pub struct SearchProjectDependency {
pub project_id: String,
pub dependency_type: DependencyType,
pub name: String,
#[serde(skip_serializing_if = "Option::is_none")]
pub slug: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub icon_url: Option<String>,
}
@@ -309,6 +346,7 @@ pub struct SearchResults {
#[derive(Serialize, Deserialize, Debug, Clone, ToSchema)]
pub struct ResultSearchProject {
/// ID of the most recently published version.
pub version_id: String,
pub project_id: String,
pub project_types: Vec<String>,