require 'open-uri' # allows to load URLs using open() require 'json' require 'deb_importer' # Search and embedding logic keeps accumulating here; the class is # cohesive enough that splitting it is not worth the indirection. # rubocop:disable Metrics/ClassLength class Package < ApplicationRecord # PostgreSQL-based full-text search: # https://github.com/Casecommons/pg_search include PgSearch::Model # TODO: Make search weighted on users' rating pg_search_scope :general_search, # :against => [:name, :description, :long_description], against: [ [:name, 'A'], [:description, 'B'], [:long_description, 'C'] ], using: { tsearch: { dictionary: 'english' } } # I am using "destroy_all" here so that when a package gets destroys the # callbacks for all dependent screenshots are executed - thus removing the # screenshot files from disk. has_many :screenshots, -> { order(id: :desc) }, inverse_of: :package, dependent: :destroy # Nearest neighbor search using pgvector. The "embedding" column stores # a vector computed from description + long_description by an external # web service (see Vectorizer and lib/tasks/vectorize_packages.rake). has_neighbors :embedding # default_scope { # order('name ASC') # } # Define the parameter(s) used for a /package/:name URL to an instance of this model def to_param name end # Return the first paragraph of the long description. def long_description_first_paragraph long_description.split(/\n\.\n/).first if long_description end # The text that the embedding vector is computed from: description # and long description combined. def embedding_text [description, long_description].compact.join("\n") end # Compute and store the embedding vector for this package by asking # the external vector service. Raises Vectorizer::Error on failure. # Returns true if a vector was saved. def update_embedding! text = embedding_text return false if text.blank? update_column(:embedding, Vectorizer.embed(text)) end # Return a relation of packages whose stored embeddings are closest to # the given query vector (an array of floats). Records get a # "neighbor_distance" attribute added (cosine distance, lower is better). # The returned relation can still be chained (e.g. filtered or paginated). # # Only returns packages that already have an embedding computed. def self.nearest_to_vector(vector, limit: nil) neighbors = nearest_neighbors(:embedding, vector, distance: 'cosine') limit ? neighbors.limit(limit) : neighbors end # Turn a free-text search string into a vector via the external web # service and find the closest packages. def self.nearest_to_text(text, limit: nil) return none if text.blank? nearest_to_vector(Vectorizer.embed(text), limit: limit) end # Packages whose name starts with the given string - used to promote # obvious matches like "sqlite3" for a "sqlite" search. def self.name_starts_with(text) return none if text.blank? where('name ILIKE ? ESCAPE ?', "#{escape_for_like(text)}%", '\\').order(:name) end # Packages whose name contains all given tokens as substrings. Catches # compound words split by the user, e.g. "sqlite browser" finding the # "sqlitebrowser" package. def self.name_contains_all(tokens) tokens = tokens.select { |token| token.length >= 2 } return none if tokens.empty? scope = self tokens.each do |token| scope = scope.where('name ILIKE ? ESCAPE ?', "%#{escape_for_like(token)}%", '\\') end scope.order(:name) end def self.escape_for_like(text) # Escape LIKE wildcards so that package name characters like "+" or "." # cannot break the pattern text.gsub(/[\\%_]/) { |char| "\\#{char}" } end # Packages semantically similar to this one, based on the stored # description embedding. Returns an empty relation for packages that # have no embedding (yet). def related_packages(limit: 6) return Package.none unless embedding Package.nearest_to_vector(embedding, limit: limit + 1) .where.not(id: id).limit(limit) end # Check whether the given text shares at least one lexeme with any # package name/description in the database. Queries without such an # overlap (gibberish, keyboard mash, unsupported languages) can # neither match full-text nor produce meaningful semantic results - # the embedding model maps unknown tokens close to the corpus # centroid, making them look like good matches. def self.lexical_overlap?(text) tokens = text.to_s.split(/\s+/).select { |t| t.match?(/[[:alpha:]]{2,}/) } return false if tokens.empty? # Pin the same text search dictionary that the packages_fts index # uses - the server-wide default can be anything. tsquery = tokens.map do |token| sanitize_sql(["plainto_tsquery('english', ?)", token]) end.join(' || ') # The expression must match the packages_fts GIN index definition # so that Postgres can use the index. where( "setweight(to_tsvector('english', coalesce(name::text, '')), 'A') || " \ "setweight(to_tsvector('english', coalesce(description::text, '')), 'B') || " \ "setweight(to_tsvector('english', coalesce(long_description, '')), 'C') @@ (#{tsquery})" ).exists? end # Return a query of all packages that have screenshots def self.with_screenshots # Query for all packages who's ID appears in a screenshot's "package_id" field subselect = Screenshot.select(:package_id) where(id: subselect) end # Return a query of all packages that have approved screenshots def self.with_public_screenshots # Query for all packages who's ID appears in a screenshot's "package_id" field subselect = Screenshot.visible.approved.select(:package_id) where(id: subselect) end # Return a list of packages that have screenshots to be moderated def self.need_moderation Package.joins(:screenshots).where('screenshots.approved=false').distinct(:name) end # Return a query of all packages that do not have screenshots def self.without_screenshots # Query for all packages who's ID does not appear in a screenshot's "package_id" field subselect = Screenshot.select(:package_id) where.not(id: subselect) end # Get all screenshots and have them sorted descendingly by their version number (Debian style). def screenshots_sorted_by_version screenshots.approved.to_a.sort { |x, y| version_compare(x.version, y.version) } end # Return all approved screenshots for this package def approved_screenshots screenshots.approved end # Return the newest screenshot that is not newer than the given version. # This algorithm collects all image # versions of a package and determines the (second) newest version. # E.g. if there are version 1.0 and 2.0 and the user is looking for # a screenshot of version 1.5 then the 1.0 version is returned. # This way the user does not see a screenshot of version 2.0 because # 2.0 might contain features that were not there in version 1.5. def best_screenshot_for_version(version) sorted_screenshots = screenshots_sorted_by_version sorted_screenshots.each do |ss| logger.debug { "Comparing version #{version} against #{ss.version}" } return ss if version_compare(version, ss.version) <= 0 end sorted_screenshots.last end # Return the part of the version up to the first - or + def upstream_version version.split(/[-+]/).first end private def version_compare(x, y) x = '0' unless x.present? y = '0' unless y.present? version_x = DebImporter::Version.new(x) version_y = DebImporter::Version.new(y) if version_x < version_y 1 elsif version_x > version_y -1 else 0 end end end # rubocop:enable Metrics/ClassLength