remove all files (#1516)
This commit is contained in:
@@ -1,101 +0,0 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
namespace :ai do
|
||||
desc "Generate topics with AI post content using random users and categories. Use it this way rake ai:generate_topics['title1\,title2\,title3']"
|
||||
task :generate_topics, [:titles] => [:environment] do |task, args|
|
||||
titles = args[:titles].include?(",") ? args[:titles].split(",").map(&:strip) : [args[:titles]]
|
||||
puts "Will create #{titles.size} #{"topics".pluralize(titles.size)}: #{titles.join(", ")}"
|
||||
|
||||
titles.each do |title|
|
||||
next if Topic.find_by_title(TextCleaner.clean_title(TextSentinel.title_sentinel(title).text))
|
||||
category =
|
||||
Category
|
||||
.where.not(id: SiteSetting.uncategorized_category_id)
|
||||
.where(read_restricted: false)
|
||||
.order("RANDOM()")
|
||||
.first
|
||||
users = User.real.activated.not_suspended.where(staged: false).order("RANDOM()").limit(12)
|
||||
RateLimiter.disable
|
||||
|
||||
creator = TopicGenerator.new(title)
|
||||
first_post = creator.get_first_post
|
||||
replies_count = rand(4..10)
|
||||
replies = creator.get_replies(replies_count)
|
||||
|
||||
post = create_topic(category, first_post, title, users)
|
||||
replies.each_with_index { |reply, i| create_post(users[i + 1], post.topic_id, reply) }
|
||||
puts "'#{title}' has #{replies.size} replies"
|
||||
end
|
||||
end
|
||||
|
||||
def create_topic(category, first_post, title, users)
|
||||
puts "#{users.first.username} will create topic '#{title}' in category '#{category.name}'"
|
||||
post =
|
||||
PostCreator.create!(
|
||||
users.first,
|
||||
title: title,
|
||||
raw: first_post,
|
||||
category: category.id,
|
||||
skip_guardian: true,
|
||||
)
|
||||
puts "Created topic '#{title}' (#{post.topic_id}) in category '#{category.name}'"
|
||||
post
|
||||
end
|
||||
|
||||
def create_post(user, topic_id, raw)
|
||||
puts "#{user.username} will reply to topic #{topic_id}"
|
||||
PostCreator.create!(user, topic_id:, raw:, skip_guardian: true)
|
||||
end
|
||||
|
||||
class TopicGenerator
|
||||
def initialize(title)
|
||||
@title = title
|
||||
end
|
||||
|
||||
def get_first_post
|
||||
TopicGenerator.generate(<<~PROMPT)
|
||||
Write and opening topic about title: #{@title}.
|
||||
- content must be in the same language as title
|
||||
- content in markdown
|
||||
- content should exclude the title
|
||||
- maximum of 200 words
|
||||
PROMPT
|
||||
end
|
||||
|
||||
def get_replies(count)
|
||||
JSON.parse(
|
||||
TopicGenerator.generate(<<~PROMPT).gsub(/```json\n?|\```/, "").gsub(/,\n\n/, ",\n").strip,
|
||||
Write #{count} replies to a topic with title #{@title}.
|
||||
- respond in an array of strings within double quotes ["", "", ""]
|
||||
- each with a maximum of 100 words
|
||||
- keep to same language of title
|
||||
- each reply may contain markdown to bold, italicize, link, or bullet point
|
||||
- do not return anything else other than the array
|
||||
- the last item in the array should not have a trailing comma
|
||||
- Example return value ["I agree with you. So and so...", "It is fun ... etc"]
|
||||
PROMPT
|
||||
)
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def self.generate(prompt)
|
||||
return "" if prompt.blank?
|
||||
|
||||
prompt =
|
||||
DiscourseAi::Completions::Prompt.new(
|
||||
"You are a forum user writing concise, informative posts. Keep responses focused and natural.",
|
||||
messages: [{ type: :user, content: prompt, id: "user" }],
|
||||
)
|
||||
|
||||
DiscourseAi::Completions::Llm.proxy(SiteSetting.ai_helper_model).generate(
|
||||
prompt,
|
||||
user: Discourse.system_user,
|
||||
feature_name: "topic-generator",
|
||||
)
|
||||
rescue => e
|
||||
Rails.logger.error("AI TopicGenerator Error: #{e.message}")
|
||||
""
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -1,29 +0,0 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
desc "Scan first posts of topics from a date, end date is optional. Usage: rake ai:spam:scan_topics[2024-01-01,2024-02-31]"
|
||||
task "ai:spam:scan_topics", %i[start_date end_date] => [:environment] do |_, args|
|
||||
start_date = args[:start_date] ? DateTime.parse(args[:start_date]) : 1.day.ago
|
||||
end_date = args[:end_date] ? DateTime.parse(args[:end_date]) : Time.current
|
||||
|
||||
scope = Topic.joins(:posts).where(created_at: start_date..end_date).where("posts.post_number = 1")
|
||||
puts "Processing #{scope.count} topics from #{start_date} to #{end_date}"
|
||||
scope
|
||||
.select("topics.id, posts.id as post_id")
|
||||
.find_each(batch_size: 500) do |record|
|
||||
Jobs.enqueue(:ai_spam_scan, post_id: record.post_id)
|
||||
print "."
|
||||
end
|
||||
end
|
||||
|
||||
desc "Scan posts from a date, end date is optional. Usage: rake ai:spam:scan_posts[2024-01-31,2024-02-01]"
|
||||
task "ai:spam:scan_posts", %i[start_date end_date] => [:environment] do |_, args|
|
||||
start_date = args[:start_date] ? DateTime.parse(args[:start_date]) : 1.day.ago
|
||||
end_date = args[:end_date] ? DateTime.parse(args[:end_date]) : Time.current
|
||||
|
||||
scope = Post.where(created_at: start_date..end_date).select(:id)
|
||||
puts "Processing #{scope.count} posts from #{start_date} to #{end_date}"
|
||||
scope.find_each(batch_size: 500) do |post|
|
||||
Jobs.enqueue(:ai_spam_scan, post_id: post.id)
|
||||
print "."
|
||||
end
|
||||
end
|
||||
@@ -1,42 +0,0 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
desc "Backfill embeddings for all topics and posts"
|
||||
task "ai:embeddings:backfill", %i[embedding_def_id concurrency] => [:environment] do |_, args|
|
||||
public_categories = Category.where(read_restricted: false).pluck(:id)
|
||||
|
||||
if args[:embedding_def_id].present?
|
||||
vdef = EmbeddingDefinition.find(args[:embedding_def_id])
|
||||
vector_rep = DiscourseAi::Embeddings::Vector.new(vdef)
|
||||
else
|
||||
vector_rep = DiscourseAi::Embeddings::Vector.instance
|
||||
end
|
||||
topics_table_name = DiscourseAi::Embeddings::Schema::TOPICS_TABLE
|
||||
|
||||
topics =
|
||||
Topic
|
||||
.joins("LEFT JOIN #{topics_table_name} ON #{topics_table_name}.topic_id = topics.id")
|
||||
.where("#{topics_table_name}.topic_id IS NULL")
|
||||
.where("category_id IN (?)", public_categories)
|
||||
.where(deleted_at: nil)
|
||||
.order("topics.id DESC")
|
||||
|
||||
Parallel.each(topics.all, in_processes: args[:concurrency].to_i, progress: "Topics") do |t|
|
||||
ActiveRecord::Base.connection_pool.with_connection do
|
||||
vector_rep.generate_representation_from(t)
|
||||
end
|
||||
end
|
||||
|
||||
posts_table_name = DiscourseAi::Embeddings::Schema::POSTS_TABLE
|
||||
posts =
|
||||
Post
|
||||
.joins("LEFT JOIN #{posts_table_name} ON #{posts_table_name}.post_id = posts.id")
|
||||
.where("#{posts_table_name}.post_id IS NULL")
|
||||
.where(deleted_at: nil)
|
||||
.order("posts.id DESC")
|
||||
|
||||
Parallel.each(posts.all, in_processes: args[:concurrency].to_i, progress: "Posts") do |t|
|
||||
ActiveRecord::Base.connection_pool.with_connection do
|
||||
vector_rep.generate_representation_from(t)
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -1,11 +0,0 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
desc "Backfill sentiment for all posts"
|
||||
task "ai:sentiment:backfill", [:start_post] => [:environment] do |_, args|
|
||||
DiscourseAi::Sentiment::PostClassification
|
||||
.backfill_query(from_post_id: args[:start_post].to_i)
|
||||
.find_in_batches do |batch|
|
||||
print "."
|
||||
DiscourseAi::Sentiment::PostClassification.new.bulk_classify!(batch)
|
||||
end
|
||||
end
|
||||
@@ -1,62 +0,0 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
desc "Creates sample sentiment / emotion data"
|
||||
task "ai:sentiment:populate", [:start_post] => [:environment] do |_, args|
|
||||
raise "Don't run this task in production!" if Rails.env.production?
|
||||
|
||||
Post
|
||||
.joins(<<~SQL)
|
||||
LEFT JOIN classification_results ON
|
||||
posts.id = classification_results.target_id AND
|
||||
classification_results.target_type = 'Post' AND
|
||||
model_used = 'cardiffnlp/twitter-roberta-base-sentiment-latest'
|
||||
SQL
|
||||
.where("classification_results.id IS NULL")
|
||||
.where("posts.id > ?", args[:start_post].to_i || 0)
|
||||
.find_each do |post|
|
||||
positive = rand(0.0..1.0)
|
||||
negative = rand(0.0..(1.0 - positive))
|
||||
neutral = 1 - positive - negative
|
||||
|
||||
ClassificationResult.create!(
|
||||
target_id: post.id,
|
||||
model_used: "cardiffnlp/twitter-roberta-base-sentiment-latest",
|
||||
classification_type: "sentiment",
|
||||
target_type: "Post",
|
||||
classification: {
|
||||
neutral: neutral,
|
||||
positive: positive,
|
||||
negative: negative,
|
||||
},
|
||||
)
|
||||
end
|
||||
|
||||
Post
|
||||
.joins(<<~SQL)
|
||||
LEFT JOIN classification_results ON
|
||||
posts.id = classification_results.target_id AND
|
||||
classification_results.target_type = 'Post' AND
|
||||
classification_results.model_used = 'SamLowe/roberta-base-go_emotions'
|
||||
SQL
|
||||
.where("classification_results.id IS NULL")
|
||||
.where("posts.id > ?", args[:start_post].to_i || 0)
|
||||
.find_each do |post|
|
||||
emotions =
|
||||
DiscourseAi::Sentiment::Emotions::LIST
|
||||
.shuffle
|
||||
.reduce({}) do |acc, emotion|
|
||||
current_sum = acc.values.sum
|
||||
acc.merge(emotion => rand(0.0..(1.0 - current_sum)))
|
||||
end
|
||||
|
||||
emotions["neutral"] = 1 - (emotions.values.sum - emotions["neutral"])
|
||||
|
||||
ClassificationResult.create!(
|
||||
target_id: post.id,
|
||||
model_used: "SamLowe/roberta-base-go_emotions",
|
||||
classification_type: "sentiment",
|
||||
target_type: "Post",
|
||||
classification: emotions,
|
||||
)
|
||||
end
|
||||
end
|
||||
@@ -1,77 +0,0 @@
|
||||
# frozen_string_literal: true
|
||||
|
||||
def classify(content)
|
||||
::DiscourseAi::Inference::DiscourseClassifier.perform!(
|
||||
"#{SiteSetting.ai_toxicity_inference_service_api_endpoint}/api/v1/classify",
|
||||
SiteSetting.ai_toxicity_inference_service_api_model,
|
||||
content,
|
||||
SiteSetting.ai_toxicity_inference_service_api_key,
|
||||
)
|
||||
end
|
||||
|
||||
desc "Uses existing flagged posts to suggest a configuration threshold"
|
||||
task "ai:toxicity:calibration_stats", [:set_size] => [:environment] do |_, args|
|
||||
flag_agreed =
|
||||
PostAction
|
||||
.where(post_action_type_id: 4, disagreed_at: nil, deferred_at: nil)
|
||||
.where("post_actions.user_id > 0")
|
||||
.includes(:post, :user)
|
||||
.where(user: { admin: false, moderator: false })
|
||||
.where("posts.raw IS NOT NULL")
|
||||
.order(created_at: :desc)
|
||||
.limit(args[:set_size])
|
||||
.pluck(:raw)
|
||||
|
||||
flag_not_agreed =
|
||||
PostAction
|
||||
.where(post_action_type_id: 4)
|
||||
.where("(disagreed_at IS NOT NULL OR deferred_at IS NOT NULL)")
|
||||
.where("post_actions.user_id > 0")
|
||||
.includes(:post, :user)
|
||||
.where(user: { admin: false, moderator: false })
|
||||
.where("posts.raw IS NOT NULL")
|
||||
.order(created_at: :desc)
|
||||
.limit(args[:set_size])
|
||||
.pluck(:raw)
|
||||
|
||||
flag_agreed_scores = flag_agreed.map { classify(_1) }
|
||||
flag_not_agreed_scores = flag_not_agreed.map { classify(_1) }
|
||||
|
||||
DiscourseAi::Toxicity::Classifier::CLASSIFICATION_LABELS.each do |label|
|
||||
puts "Label: #{label}"
|
||||
|
||||
label_agreed_scores = flag_agreed_scores.map { _1[label] }
|
||||
label_not_agreed_scores = flag_not_agreed_scores.map { _1[label] }
|
||||
|
||||
puts "Flagged posts score:"
|
||||
puts "Max: #{label_agreed_scores.max}"
|
||||
puts "Min: #{label_agreed_scores.min}"
|
||||
puts "Avg: #{label_agreed_scores.sum(0.0) / label_agreed_scores.size}"
|
||||
puts "Median: #{label_agreed_scores.sort[label_agreed_scores.size / 2]}"
|
||||
puts "Stddev: #{Math.sqrt(label_agreed_scores.map { (_1 - label_agreed_scores.sum(0.0) / label_agreed_scores.size)**2 }.sum(0.0) / label_agreed_scores.size)}"
|
||||
|
||||
puts "Flagged posts score:"
|
||||
puts "Max: #{label_not_agreed_scores.max}"
|
||||
puts "Min: #{label_not_agreed_scores.min}"
|
||||
puts "Avg: #{label_not_agreed_scores.sum(0.0) / label_not_agreed_scores.size}"
|
||||
puts "Median: #{label_not_agreed_scores.sort[label_not_agreed_scores.size / 2]}"
|
||||
puts "Stddev: #{Math.sqrt(label_not_agreed_scores.map { (_1 - label_not_agreed_scores.sum(0.0) / label_not_agreed_scores.size)**2 }.sum(0.0) / label_not_agreed_scores.size)}"
|
||||
|
||||
best_cutoff = 0
|
||||
best_cutoff_score = 0
|
||||
|
||||
(0..100)
|
||||
.step(1)
|
||||
.each do |cutoff|
|
||||
score =
|
||||
label_agreed_scores.count { _1 > cutoff } + label_not_agreed_scores.count { _1 <= cutoff }
|
||||
|
||||
if score > best_cutoff_score
|
||||
best_cutoff_score = score
|
||||
best_cutoff = cutoff
|
||||
end
|
||||
end
|
||||
|
||||
puts "Recommended ai_toxicity_flag_threshold_#{label} value: #{best_cutoff}"
|
||||
end
|
||||
end
|
||||
Reference in New Issue
Block a user