debshots/app/models/package.rb
Christoph Haas 4948fe50e3 Add semantic package search via pgvector embeddings
Package text search now combines several strategies: exact name match,
name prefix promotion, compound word splitting ("sqlite browser" finds
"sqlitebrowser") and semantic nearest neighbor search over description
embeddings computed by an external embedding service (all-MiniLM-L6-v2,
384 dims) stored with the pgvector extension. Classic PostgreSQL
full-text search remains as fallback when the vector service is
unreachable. Gibberish queries without lexical overlap with the package
data return empty results instead of random matches.

Embeddings can be backfilled with bin/rails debshots:compute_vectors.

Also drops the unused lograge gem from the Gemfile.
2026-08-22 22:07:56 +02:00

211 lines
7.5 KiB
Ruby

require 'open-uri' # allows to load URLs using open()
require 'json'
require 'deb_importer'
# Search and embedding logic keeps accumulating here; the class is
# cohesive enough that splitting it is not worth the indirection.
# rubocop:disable Metrics/ClassLength
class Package < ApplicationRecord
# PostgreSQL-based full-text search:
# https://github.com/Casecommons/pg_search
include PgSearch::Model
# TODO: Make search weighted on users' rating
pg_search_scope :general_search,
# :against => [:name, :description, :long_description],
against: [
[:name, 'A'],
[:description, 'B'],
[:long_description, 'C']
],
using: {
tsearch: { dictionary: 'english' }
}
# I am using "destroy_all" here so that when a package gets destroys the
# callbacks for all dependent screenshots are executed - thus removing the
# screenshot files from disk.
has_many :screenshots,
-> { order(id: :desc) },
inverse_of: :package,
dependent: :destroy
# Nearest neighbor search using pgvector. The "embedding" column stores
# a vector computed from description + long_description by an external
# web service (see Vectorizer and lib/tasks/vectorize_packages.rake).
has_neighbors :embedding
# default_scope {
# order('name ASC')
# }
# Define the parameter(s) used for a /package/:name URL to an instance of this model
def to_param
name
end
# Return the first paragraph of the long description.
def long_description_first_paragraph
long_description.split(/\n\.\n/).first if long_description
end
# The text that the embedding vector is computed from: description
# and long description combined.
def embedding_text
[description, long_description].compact.join("\n")
end
# Compute and store the embedding vector for this package by asking
# the external vector service. Raises Vectorizer::Error on failure.
# Returns true if the vector was saved.
def update_embedding!
text = embedding_text
return false if text.blank?
self.embedding = Vectorizer.embed(text)
save!
end
# Return a relation of packages whose stored embeddings are closest to
# the given query vector (an array of floats). Records get a
# "neighbor_distance" attribute added (cosine distance, lower is better).
# The returned relation can still be chained (e.g. filtered or paginated).
#
# Only returns packages that already have an embedding computed.
def self.nearest_to_vector(vector, limit: nil)
neighbors = nearest_neighbors(:embedding, vector, distance: 'cosine')
limit ? neighbors.limit(limit) : neighbors
end
# Turn a free-text search string into a vector via the external web
# service and find the closest packages.
def self.nearest_to_text(text, limit: nil)
return none if text.blank?
nearest_to_vector(Vectorizer.embed(text), limit: limit)
end
# Packages whose name starts with the given string - used to promote
# obvious matches like "sqlite3" for a "sqlite" search.
def self.name_starts_with(text)
return none if text.blank?
where('name ILIKE ? ESCAPE ?', "#{escape_for_like(text)}%", '\\').order(:name)
end
# Packages whose name contains all given tokens as substrings. Catches
# compound words split by the user, e.g. "sqlite browser" finding the
# "sqlitebrowser" package.
def self.name_contains_all(tokens)
tokens = tokens.select { |token| token.length >= 2 }
return none if tokens.empty?
scope = self
tokens.each do |token|
scope = scope.where('name ILIKE ? ESCAPE ?', "%#{escape_for_like(token)}%", '\\')
end
scope.order(:name)
end
def self.escape_for_like(text)
# Escape LIKE wildcards so that package name characters like "+" or "."
# cannot break the pattern
text.gsub(/[\\%_]/) { |char| "\\#{char}" }
end
# Check whether the given text shares at least one lexeme with any
# package name/description in the database. Queries without such an
# overlap (gibberish, keyboard mash, unsupported languages) can
# neither match full-text nor produce meaningful semantic results -
# the embedding model maps unknown tokens close to the corpus
# centroid, making them look like good matches.
def self.lexical_overlap?(text)
tokens = text.to_s.split(/\s+/).select { |t| t.match?(/[[:alpha:]]{2,}/) }
return false if tokens.empty?
# Pin the same text search dictionary that the packages_fts index
# uses - the server-wide default can be anything.
tsquery = tokens.map do |token|
sanitize_sql(["plainto_tsquery('english', ?)", token])
end.join(' || ')
# The expression must match the packages_fts GIN index definition
# so that Postgres can use the index.
where(
"setweight(to_tsvector('english', coalesce(name::text, '')), 'A') || " \
"setweight(to_tsvector('english', coalesce(description::text, '')), 'B') || " \
"setweight(to_tsvector('english', coalesce(long_description, '')), 'C') @@ (#{tsquery})"
).exists?
end
# Return a query of all packages that have screenshots
def self.with_screenshots
# Query for all packages who's ID appears in a screenshot's "package_id" field
subselect = Screenshot.select(:package_id)
where(id: subselect)
end
# Return a query of all packages that have approved screenshots
def self.with_public_screenshots
# Query for all packages who's ID appears in a screenshot's "package_id" field
subselect = Screenshot.visible.approved.select(:package_id)
where(id: subselect)
end
# Return a list of packages that have screenshots to be moderated
def self.need_moderation
Package.joins(:screenshots).where('screenshots.approved=false').distinct(:name)
end
# Return a query of all packages that do not have screenshots
def self.without_screenshots
# Query for all packages who's ID does not appear in a screenshot's "package_id" field
subselect = Screenshot.select(:package_id)
where.not(id: subselect)
end
# Get all screenshots and have them sorted descendingly by their version number (Debian style).
def screenshots_sorted_by_version
screenshots.approved.to_a.sort { |x, y| version_compare(x.version, y.version) }
end
# Return all approved screenshots for this package
def approved_screenshots
screenshots.approved
end
# Return the newest screenshot that is not newer than the given version.
# This algorithm collects all image
# versions of a package and determines the (second) newest version.
# E.g. if there are version 1.0 and 2.0 and the user is looking for
# a screenshot of version 1.5 then the 1.0 version is returned.
# This way the user does not see a screenshot of version 2.0 because
# 2.0 might contain features that were not there in version 1.5.
def best_screenshot_for_version(version)
sorted_screenshots = screenshots_sorted_by_version
sorted_screenshots.each do |ss|
logger.debug { "Comparing version #{version} against #{ss.version}" }
return ss if version_compare(version, ss.version) <= 0
end
sorted_screenshots.last
end
# Return the part of the version up to the first - or +
def upstream_version
version.split(/[-+]/).first
end
private
def version_compare(x, y)
x = '0' unless x.present?
y = '0' unless y.present?
version_x = DebImporter::Version.new(x)
version_y = DebImporter::Version.new(y)
if version_x < version_y
1
elsif version_x > version_y
-1
else
0
end
end
end