debshots/app/models/package.rb
Christoph Haas 3550a4e2e2 Harden vectorization task against per-package failures
A single failing package used to kill its whole worker thread silently
because only Vectorizer::Error was rescued per package. Now every
StandardError is caught and logged with class and backtrace, and the
embedding write is narrowed to update_column(:embedding) so the task
only ever issues a minimal UPDATE - it cannot touch any other column
or fire model callbacks. Progress stats no longer serialize database
writes through the mutex.
2026-08-23 00:11:52 +02:00

221 lines
7.9 KiB
Ruby

require 'open-uri' # allows to load URLs using open()
require 'json'
require 'deb_importer'
# Search and embedding logic keeps accumulating here; the class is
# cohesive enough that splitting it is not worth the indirection.
# rubocop:disable Metrics/ClassLength
class Package < ApplicationRecord
# PostgreSQL-based full-text search:
# https://github.com/Casecommons/pg_search
include PgSearch::Model
# TODO: Make search weighted on users' rating
pg_search_scope :general_search,
# :against => [:name, :description, :long_description],
against: [
[:name, 'A'],
[:description, 'B'],
[:long_description, 'C']
],
using: {
tsearch: { dictionary: 'english' }
}
# I am using "destroy_all" here so that when a package gets destroys the
# callbacks for all dependent screenshots are executed - thus removing the
# screenshot files from disk.
has_many :screenshots,
-> { order(id: :desc) },
inverse_of: :package,
dependent: :destroy
# Nearest neighbor search using pgvector. The "embedding" column stores
# a vector computed from description + long_description by an external
# web service (see Vectorizer and lib/tasks/vectorize_packages.rake).
has_neighbors :embedding
# default_scope {
# order('name ASC')
# }
# Define the parameter(s) used for a /package/:name URL to an instance of this model
def to_param
name
end
# Return the first paragraph of the long description.
def long_description_first_paragraph
long_description.split(/\n\.\n/).first if long_description
end
# The text that the embedding vector is computed from: description
# and long description combined.
def embedding_text
[description, long_description].compact.join("\n")
end
# Compute and store the embedding vector for this package by asking
# the external vector service. Raises Vectorizer::Error on failure.
# Returns true if a vector was saved.
def update_embedding!
text = embedding_text
return false if text.blank?
update_column(:embedding, Vectorizer.embed(text))
end
# Return a relation of packages whose stored embeddings are closest to
# the given query vector (an array of floats). Records get a
# "neighbor_distance" attribute added (cosine distance, lower is better).
# The returned relation can still be chained (e.g. filtered or paginated).
#
# Only returns packages that already have an embedding computed.
def self.nearest_to_vector(vector, limit: nil)
neighbors = nearest_neighbors(:embedding, vector, distance: 'cosine')
limit ? neighbors.limit(limit) : neighbors
end
# Turn a free-text search string into a vector via the external web
# service and find the closest packages.
def self.nearest_to_text(text, limit: nil)
return none if text.blank?
nearest_to_vector(Vectorizer.embed(text), limit: limit)
end
# Packages whose name starts with the given string - used to promote
# obvious matches like "sqlite3" for a "sqlite" search.
def self.name_starts_with(text)
return none if text.blank?
where('name ILIKE ? ESCAPE ?', "#{escape_for_like(text)}%", '\\').order(:name)
end
# Packages whose name contains all given tokens as substrings. Catches
# compound words split by the user, e.g. "sqlite browser" finding the
# "sqlitebrowser" package.
def self.name_contains_all(tokens)
tokens = tokens.select { |token| token.length >= 2 }
return none if tokens.empty?
scope = self
tokens.each do |token|
scope = scope.where('name ILIKE ? ESCAPE ?', "%#{escape_for_like(token)}%", '\\')
end
scope.order(:name)
end
def self.escape_for_like(text)
# Escape LIKE wildcards so that package name characters like "+" or "."
# cannot break the pattern
text.gsub(/[\\%_]/) { |char| "\\#{char}" }
end
# Packages semantically similar to this one, based on the stored
# description embedding. Returns an empty relation for packages that
# have no embedding (yet).
def related_packages(limit: 6)
return Package.none unless embedding
Package.nearest_to_vector(embedding, limit: limit + 1)
.where.not(id: id).limit(limit)
end
# Check whether the given text shares at least one lexeme with any
# package name/description in the database. Queries without such an
# overlap (gibberish, keyboard mash, unsupported languages) can
# neither match full-text nor produce meaningful semantic results -
# the embedding model maps unknown tokens close to the corpus
# centroid, making them look like good matches.
def self.lexical_overlap?(text)
tokens = text.to_s.split(/\s+/).select { |t| t.match?(/[[:alpha:]]{2,}/) }
return false if tokens.empty?
# Pin the same text search dictionary that the packages_fts index
# uses - the server-wide default can be anything.
tsquery = tokens.map do |token|
sanitize_sql(["plainto_tsquery('english', ?)", token])
end.join(' || ')
# The expression must match the packages_fts GIN index definition
# so that Postgres can use the index.
where(
"setweight(to_tsvector('english', coalesce(name::text, '')), 'A') || " \
"setweight(to_tsvector('english', coalesce(description::text, '')), 'B') || " \
"setweight(to_tsvector('english', coalesce(long_description, '')), 'C') @@ (#{tsquery})"
).exists?
end
# Return a query of all packages that have screenshots
def self.with_screenshots
# Query for all packages who's ID appears in a screenshot's "package_id" field
subselect = Screenshot.select(:package_id)
where(id: subselect)
end
# Return a query of all packages that have approved screenshots
def self.with_public_screenshots
# Query for all packages who's ID appears in a screenshot's "package_id" field
subselect = Screenshot.visible.approved.select(:package_id)
where(id: subselect)
end
# Return a list of packages that have screenshots to be moderated
def self.need_moderation
Package.joins(:screenshots).where('screenshots.approved=false').distinct(:name)
end
# Return a query of all packages that do not have screenshots
def self.without_screenshots
# Query for all packages who's ID does not appear in a screenshot's "package_id" field
subselect = Screenshot.select(:package_id)
where.not(id: subselect)
end
# Get all screenshots and have them sorted descendingly by their version number (Debian style).
def screenshots_sorted_by_version
screenshots.approved.to_a.sort { |x, y| version_compare(x.version, y.version) }
end
# Return all approved screenshots for this package
def approved_screenshots
screenshots.approved
end
# Return the newest screenshot that is not newer than the given version.
# This algorithm collects all image
# versions of a package and determines the (second) newest version.
# E.g. if there are version 1.0 and 2.0 and the user is looking for
# a screenshot of version 1.5 then the 1.0 version is returned.
# This way the user does not see a screenshot of version 2.0 because
# 2.0 might contain features that were not there in version 1.5.
def best_screenshot_for_version(version)
sorted_screenshots = screenshots_sorted_by_version
sorted_screenshots.each do |ss|
logger.debug { "Comparing version #{version} against #{ss.version}" }
return ss if version_compare(version, ss.version) <= 0
end
sorted_screenshots.last
end
# Return the part of the version up to the first - or +
def upstream_version
version.split(/[-+]/).first
end
private
def version_compare(x, y)
x = '0' unless x.present?
y = '0' unless y.present?
version_x = DebImporter::Version.new(x)
version_y = DebImporter::Version.new(y)
if version_x < version_y
1
elsif version_x > version_y
-1
else
0
end
end
end
# rubocop:enable Metrics/ClassLength