Package text search now combines several strategies: exact name match,
name prefix promotion, compound word splitting ("sqlite browser" finds
"sqlitebrowser") and semantic nearest neighbor search over description
embeddings computed by an external embedding service (all-MiniLM-L6-v2,
384 dims) stored with the pgvector extension. Classic PostgreSQL
full-text search remains as fallback when the vector service is
unreachable. Gibberish queries without lexical overlap with the package
data return empty results instead of random matches.
Embeddings can be backfilled with bin/rails debshots:compute_vectors.
Also drops the unused lograge gem from the Gemfile.
211 lines
7.5 KiB
Ruby
211 lines
7.5 KiB
Ruby
require 'open-uri' # allows to load URLs using open()
|
|
require 'json'
|
|
require 'deb_importer'
|
|
|
|
# Search and embedding logic keeps accumulating here; the class is
|
|
# cohesive enough that splitting it is not worth the indirection.
|
|
# rubocop:disable Metrics/ClassLength
|
|
class Package < ApplicationRecord
|
|
# PostgreSQL-based full-text search:
|
|
# https://github.com/Casecommons/pg_search
|
|
include PgSearch::Model
|
|
# TODO: Make search weighted on users' rating
|
|
pg_search_scope :general_search,
|
|
# :against => [:name, :description, :long_description],
|
|
against: [
|
|
[:name, 'A'],
|
|
[:description, 'B'],
|
|
[:long_description, 'C']
|
|
],
|
|
using: {
|
|
tsearch: { dictionary: 'english' }
|
|
}
|
|
|
|
# I am using "destroy_all" here so that when a package gets destroys the
|
|
# callbacks for all dependent screenshots are executed - thus removing the
|
|
# screenshot files from disk.
|
|
has_many :screenshots,
|
|
-> { order(id: :desc) },
|
|
inverse_of: :package,
|
|
dependent: :destroy
|
|
|
|
# Nearest neighbor search using pgvector. The "embedding" column stores
|
|
# a vector computed from description + long_description by an external
|
|
# web service (see Vectorizer and lib/tasks/vectorize_packages.rake).
|
|
has_neighbors :embedding
|
|
|
|
# default_scope {
|
|
# order('name ASC')
|
|
# }
|
|
|
|
# Define the parameter(s) used for a /package/:name URL to an instance of this model
|
|
def to_param
|
|
name
|
|
end
|
|
|
|
# Return the first paragraph of the long description.
|
|
def long_description_first_paragraph
|
|
long_description.split(/\n\.\n/).first if long_description
|
|
end
|
|
|
|
# The text that the embedding vector is computed from: description
|
|
# and long description combined.
|
|
def embedding_text
|
|
[description, long_description].compact.join("\n")
|
|
end
|
|
|
|
# Compute and store the embedding vector for this package by asking
|
|
# the external vector service. Raises Vectorizer::Error on failure.
|
|
# Returns true if the vector was saved.
|
|
def update_embedding!
|
|
text = embedding_text
|
|
return false if text.blank?
|
|
|
|
self.embedding = Vectorizer.embed(text)
|
|
save!
|
|
end
|
|
|
|
# Return a relation of packages whose stored embeddings are closest to
|
|
# the given query vector (an array of floats). Records get a
|
|
# "neighbor_distance" attribute added (cosine distance, lower is better).
|
|
# The returned relation can still be chained (e.g. filtered or paginated).
|
|
#
|
|
# Only returns packages that already have an embedding computed.
|
|
def self.nearest_to_vector(vector, limit: nil)
|
|
neighbors = nearest_neighbors(:embedding, vector, distance: 'cosine')
|
|
limit ? neighbors.limit(limit) : neighbors
|
|
end
|
|
|
|
# Turn a free-text search string into a vector via the external web
|
|
# service and find the closest packages.
|
|
def self.nearest_to_text(text, limit: nil)
|
|
return none if text.blank?
|
|
|
|
nearest_to_vector(Vectorizer.embed(text), limit: limit)
|
|
end
|
|
|
|
# Packages whose name starts with the given string - used to promote
|
|
# obvious matches like "sqlite3" for a "sqlite" search.
|
|
def self.name_starts_with(text)
|
|
return none if text.blank?
|
|
|
|
where('name ILIKE ? ESCAPE ?', "#{escape_for_like(text)}%", '\\').order(:name)
|
|
end
|
|
|
|
# Packages whose name contains all given tokens as substrings. Catches
|
|
# compound words split by the user, e.g. "sqlite browser" finding the
|
|
# "sqlitebrowser" package.
|
|
def self.name_contains_all(tokens)
|
|
tokens = tokens.select { |token| token.length >= 2 }
|
|
return none if tokens.empty?
|
|
|
|
scope = self
|
|
tokens.each do |token|
|
|
scope = scope.where('name ILIKE ? ESCAPE ?', "%#{escape_for_like(token)}%", '\\')
|
|
end
|
|
scope.order(:name)
|
|
end
|
|
|
|
def self.escape_for_like(text)
|
|
# Escape LIKE wildcards so that package name characters like "+" or "."
|
|
# cannot break the pattern
|
|
text.gsub(/[\\%_]/) { |char| "\\#{char}" }
|
|
end
|
|
|
|
# Check whether the given text shares at least one lexeme with any
|
|
# package name/description in the database. Queries without such an
|
|
# overlap (gibberish, keyboard mash, unsupported languages) can
|
|
# neither match full-text nor produce meaningful semantic results -
|
|
# the embedding model maps unknown tokens close to the corpus
|
|
# centroid, making them look like good matches.
|
|
def self.lexical_overlap?(text)
|
|
tokens = text.to_s.split(/\s+/).select { |t| t.match?(/[[:alpha:]]{2,}/) }
|
|
return false if tokens.empty?
|
|
|
|
# Pin the same text search dictionary that the packages_fts index
|
|
# uses - the server-wide default can be anything.
|
|
tsquery = tokens.map do |token|
|
|
sanitize_sql(["plainto_tsquery('english', ?)", token])
|
|
end.join(' || ')
|
|
# The expression must match the packages_fts GIN index definition
|
|
# so that Postgres can use the index.
|
|
where(
|
|
"setweight(to_tsvector('english', coalesce(name::text, '')), 'A') || " \
|
|
"setweight(to_tsvector('english', coalesce(description::text, '')), 'B') || " \
|
|
"setweight(to_tsvector('english', coalesce(long_description, '')), 'C') @@ (#{tsquery})"
|
|
).exists?
|
|
end
|
|
|
|
# Return a query of all packages that have screenshots
|
|
def self.with_screenshots
|
|
# Query for all packages who's ID appears in a screenshot's "package_id" field
|
|
subselect = Screenshot.select(:package_id)
|
|
where(id: subselect)
|
|
end
|
|
|
|
# Return a query of all packages that have approved screenshots
|
|
def self.with_public_screenshots
|
|
# Query for all packages who's ID appears in a screenshot's "package_id" field
|
|
subselect = Screenshot.visible.approved.select(:package_id)
|
|
where(id: subselect)
|
|
end
|
|
|
|
# Return a list of packages that have screenshots to be moderated
|
|
def self.need_moderation
|
|
Package.joins(:screenshots).where('screenshots.approved=false').distinct(:name)
|
|
end
|
|
|
|
# Return a query of all packages that do not have screenshots
|
|
def self.without_screenshots
|
|
# Query for all packages who's ID does not appear in a screenshot's "package_id" field
|
|
subselect = Screenshot.select(:package_id)
|
|
where.not(id: subselect)
|
|
end
|
|
|
|
# Get all screenshots and have them sorted descendingly by their version number (Debian style).
|
|
def screenshots_sorted_by_version
|
|
screenshots.approved.to_a.sort { |x, y| version_compare(x.version, y.version) }
|
|
end
|
|
|
|
# Return all approved screenshots for this package
|
|
def approved_screenshots
|
|
screenshots.approved
|
|
end
|
|
|
|
# Return the newest screenshot that is not newer than the given version.
|
|
# This algorithm collects all image
|
|
# versions of a package and determines the (second) newest version.
|
|
# E.g. if there are version 1.0 and 2.0 and the user is looking for
|
|
# a screenshot of version 1.5 then the 1.0 version is returned.
|
|
# This way the user does not see a screenshot of version 2.0 because
|
|
# 2.0 might contain features that were not there in version 1.5.
|
|
def best_screenshot_for_version(version)
|
|
sorted_screenshots = screenshots_sorted_by_version
|
|
sorted_screenshots.each do |ss|
|
|
logger.debug { "Comparing version #{version} against #{ss.version}" }
|
|
return ss if version_compare(version, ss.version) <= 0
|
|
end
|
|
sorted_screenshots.last
|
|
end
|
|
|
|
# Return the part of the version up to the first - or +
|
|
def upstream_version
|
|
version.split(/[-+]/).first
|
|
end
|
|
|
|
private
|
|
|
|
def version_compare(x, y)
|
|
x = '0' unless x.present?
|
|
y = '0' unless y.present?
|
|
version_x = DebImporter::Version.new(x)
|
|
version_y = DebImporter::Version.new(y)
|
|
if version_x < version_y
|
|
1
|
|
elsif version_x > version_y
|
|
-1
|
|
else
|
|
0
|
|
end
|
|
end
|
|
end
|