remove all files (#1516)

This commit is contained in:
Sam
2026-08-28 08:24:24 +10:00
committed by GitHub
parent c312b3184a
commit 2c4e5e998e
1110 changed files with 2 additions and 133260 deletions
-101
View File
@@ -1,101 +0,0 @@
# frozen_string_literal: true
namespace :ai do
desc "Generate topics with AI post content using random users and categories. Use it this way rake ai:generate_topics['title1\,title2\,title3']"
task :generate_topics, [:titles] => [:environment] do |task, args|
titles = args[:titles].include?(",") ? args[:titles].split(",").map(&:strip) : [args[:titles]]
puts "Will create #{titles.size} #{"topics".pluralize(titles.size)}: #{titles.join(", ")}"
titles.each do |title|
next if Topic.find_by_title(TextCleaner.clean_title(TextSentinel.title_sentinel(title).text))
category =
Category
.where.not(id: SiteSetting.uncategorized_category_id)
.where(read_restricted: false)
.order("RANDOM()")
.first
users = User.real.activated.not_suspended.where(staged: false).order("RANDOM()").limit(12)
RateLimiter.disable
creator = TopicGenerator.new(title)
first_post = creator.get_first_post
replies_count = rand(4..10)
replies = creator.get_replies(replies_count)
post = create_topic(category, first_post, title, users)
replies.each_with_index { |reply, i| create_post(users[i + 1], post.topic_id, reply) }
puts "'#{title}' has #{replies.size} replies"
end
end
def create_topic(category, first_post, title, users)
puts "#{users.first.username} will create topic '#{title}' in category '#{category.name}'"
post =
PostCreator.create!(
users.first,
title: title,
raw: first_post,
category: category.id,
skip_guardian: true,
)
puts "Created topic '#{title}' (#{post.topic_id}) in category '#{category.name}'"
post
end
def create_post(user, topic_id, raw)
puts "#{user.username} will reply to topic #{topic_id}"
PostCreator.create!(user, topic_id:, raw:, skip_guardian: true)
end
class TopicGenerator
def initialize(title)
@title = title
end
def get_first_post
TopicGenerator.generate(<<~PROMPT)
Write and opening topic about title: #{@title}.
- content must be in the same language as title
- content in markdown
- content should exclude the title
- maximum of 200 words
PROMPT
end
def get_replies(count)
JSON.parse(
TopicGenerator.generate(<<~PROMPT).gsub(/```json\n?|\```/, "").gsub(/,\n\n/, ",\n").strip,
Write #{count} replies to a topic with title #{@title}.
- respond in an array of strings within double quotes ["", "", ""]
- each with a maximum of 100 words
- keep to same language of title
- each reply may contain markdown to bold, italicize, link, or bullet point
- do not return anything else other than the array
- the last item in the array should not have a trailing comma
- Example return value ["I agree with you. So and so...", "It is fun ... etc"]
PROMPT
)
end
private
def self.generate(prompt)
return "" if prompt.blank?
prompt =
DiscourseAi::Completions::Prompt.new(
"You are a forum user writing concise, informative posts. Keep responses focused and natural.",
messages: [{ type: :user, content: prompt, id: "user" }],
)
DiscourseAi::Completions::Llm.proxy(SiteSetting.ai_helper_model).generate(
prompt,
user: Discourse.system_user,
feature_name: "topic-generator",
)
rescue => e
Rails.logger.error("AI TopicGenerator Error: #{e.message}")
""
end
end
end
-29
View File
@@ -1,29 +0,0 @@
# frozen_string_literal: true
desc "Scan first posts of topics from a date, end date is optional. Usage: rake ai:spam:scan_topics[2024-01-01,2024-02-31]"
task "ai:spam:scan_topics", %i[start_date end_date] => [:environment] do |_, args|
start_date = args[:start_date] ? DateTime.parse(args[:start_date]) : 1.day.ago
end_date = args[:end_date] ? DateTime.parse(args[:end_date]) : Time.current
scope = Topic.joins(:posts).where(created_at: start_date..end_date).where("posts.post_number = 1")
puts "Processing #{scope.count} topics from #{start_date} to #{end_date}"
scope
.select("topics.id, posts.id as post_id")
.find_each(batch_size: 500) do |record|
Jobs.enqueue(:ai_spam_scan, post_id: record.post_id)
print "."
end
end
desc "Scan posts from a date, end date is optional. Usage: rake ai:spam:scan_posts[2024-01-31,2024-02-01]"
task "ai:spam:scan_posts", %i[start_date end_date] => [:environment] do |_, args|
start_date = args[:start_date] ? DateTime.parse(args[:start_date]) : 1.day.ago
end_date = args[:end_date] ? DateTime.parse(args[:end_date]) : Time.current
scope = Post.where(created_at: start_date..end_date).select(:id)
puts "Processing #{scope.count} posts from #{start_date} to #{end_date}"
scope.find_each(batch_size: 500) do |post|
Jobs.enqueue(:ai_spam_scan, post_id: post.id)
print "."
end
end
@@ -1,42 +0,0 @@
# frozen_string_literal: true
desc "Backfill embeddings for all topics and posts"
task "ai:embeddings:backfill", %i[embedding_def_id concurrency] => [:environment] do |_, args|
public_categories = Category.where(read_restricted: false).pluck(:id)
if args[:embedding_def_id].present?
vdef = EmbeddingDefinition.find(args[:embedding_def_id])
vector_rep = DiscourseAi::Embeddings::Vector.new(vdef)
else
vector_rep = DiscourseAi::Embeddings::Vector.instance
end
topics_table_name = DiscourseAi::Embeddings::Schema::TOPICS_TABLE
topics =
Topic
.joins("LEFT JOIN #{topics_table_name} ON #{topics_table_name}.topic_id = topics.id")
.where("#{topics_table_name}.topic_id IS NULL")
.where("category_id IN (?)", public_categories)
.where(deleted_at: nil)
.order("topics.id DESC")
Parallel.each(topics.all, in_processes: args[:concurrency].to_i, progress: "Topics") do |t|
ActiveRecord::Base.connection_pool.with_connection do
vector_rep.generate_representation_from(t)
end
end
posts_table_name = DiscourseAi::Embeddings::Schema::POSTS_TABLE
posts =
Post
.joins("LEFT JOIN #{posts_table_name} ON #{posts_table_name}.post_id = posts.id")
.where("#{posts_table_name}.post_id IS NULL")
.where(deleted_at: nil)
.order("posts.id DESC")
Parallel.each(posts.all, in_processes: args[:concurrency].to_i, progress: "Posts") do |t|
ActiveRecord::Base.connection_pool.with_connection do
vector_rep.generate_representation_from(t)
end
end
end
-11
View File
@@ -1,11 +0,0 @@
# frozen_string_literal: true
desc "Backfill sentiment for all posts"
task "ai:sentiment:backfill", [:start_post] => [:environment] do |_, args|
DiscourseAi::Sentiment::PostClassification
.backfill_query(from_post_id: args[:start_post].to_i)
.find_in_batches do |batch|
print "."
DiscourseAi::Sentiment::PostClassification.new.bulk_classify!(batch)
end
end
-62
View File
@@ -1,62 +0,0 @@
# frozen_string_literal: true
desc "Creates sample sentiment / emotion data"
task "ai:sentiment:populate", [:start_post] => [:environment] do |_, args|
raise "Don't run this task in production!" if Rails.env.production?
Post
.joins(<<~SQL)
LEFT JOIN classification_results ON
posts.id = classification_results.target_id AND
classification_results.target_type = 'Post' AND
model_used = 'cardiffnlp/twitter-roberta-base-sentiment-latest'
SQL
.where("classification_results.id IS NULL")
.where("posts.id > ?", args[:start_post].to_i || 0)
.find_each do |post|
positive = rand(0.0..1.0)
negative = rand(0.0..(1.0 - positive))
neutral = 1 - positive - negative
ClassificationResult.create!(
target_id: post.id,
model_used: "cardiffnlp/twitter-roberta-base-sentiment-latest",
classification_type: "sentiment",
target_type: "Post",
classification: {
neutral: neutral,
positive: positive,
negative: negative,
},
)
end
Post
.joins(<<~SQL)
LEFT JOIN classification_results ON
posts.id = classification_results.target_id AND
classification_results.target_type = 'Post' AND
classification_results.model_used = 'SamLowe/roberta-base-go_emotions'
SQL
.where("classification_results.id IS NULL")
.where("posts.id > ?", args[:start_post].to_i || 0)
.find_each do |post|
emotions =
DiscourseAi::Sentiment::Emotions::LIST
.shuffle
.reduce({}) do |acc, emotion|
current_sum = acc.values.sum
acc.merge(emotion => rand(0.0..(1.0 - current_sum)))
end
emotions["neutral"] = 1 - (emotions.values.sum - emotions["neutral"])
ClassificationResult.create!(
target_id: post.id,
model_used: "SamLowe/roberta-base-go_emotions",
classification_type: "sentiment",
target_type: "Post",
classification: emotions,
)
end
end
@@ -1,77 +0,0 @@
# frozen_string_literal: true
def classify(content)
::DiscourseAi::Inference::DiscourseClassifier.perform!(
"#{SiteSetting.ai_toxicity_inference_service_api_endpoint}/api/v1/classify",
SiteSetting.ai_toxicity_inference_service_api_model,
content,
SiteSetting.ai_toxicity_inference_service_api_key,
)
end
desc "Uses existing flagged posts to suggest a configuration threshold"
task "ai:toxicity:calibration_stats", [:set_size] => [:environment] do |_, args|
flag_agreed =
PostAction
.where(post_action_type_id: 4, disagreed_at: nil, deferred_at: nil)
.where("post_actions.user_id > 0")
.includes(:post, :user)
.where(user: { admin: false, moderator: false })
.where("posts.raw IS NOT NULL")
.order(created_at: :desc)
.limit(args[:set_size])
.pluck(:raw)
flag_not_agreed =
PostAction
.where(post_action_type_id: 4)
.where("(disagreed_at IS NOT NULL OR deferred_at IS NOT NULL)")
.where("post_actions.user_id > 0")
.includes(:post, :user)
.where(user: { admin: false, moderator: false })
.where("posts.raw IS NOT NULL")
.order(created_at: :desc)
.limit(args[:set_size])
.pluck(:raw)
flag_agreed_scores = flag_agreed.map { classify(_1) }
flag_not_agreed_scores = flag_not_agreed.map { classify(_1) }
DiscourseAi::Toxicity::Classifier::CLASSIFICATION_LABELS.each do |label|
puts "Label: #{label}"
label_agreed_scores = flag_agreed_scores.map { _1[label] }
label_not_agreed_scores = flag_not_agreed_scores.map { _1[label] }
puts "Flagged posts score:"
puts "Max: #{label_agreed_scores.max}"
puts "Min: #{label_agreed_scores.min}"
puts "Avg: #{label_agreed_scores.sum(0.0) / label_agreed_scores.size}"
puts "Median: #{label_agreed_scores.sort[label_agreed_scores.size / 2]}"
puts "Stddev: #{Math.sqrt(label_agreed_scores.map { (_1 - label_agreed_scores.sum(0.0) / label_agreed_scores.size)**2 }.sum(0.0) / label_agreed_scores.size)}"
puts "Flagged posts score:"
puts "Max: #{label_not_agreed_scores.max}"
puts "Min: #{label_not_agreed_scores.min}"
puts "Avg: #{label_not_agreed_scores.sum(0.0) / label_not_agreed_scores.size}"
puts "Median: #{label_not_agreed_scores.sort[label_not_agreed_scores.size / 2]}"
puts "Stddev: #{Math.sqrt(label_not_agreed_scores.map { (_1 - label_not_agreed_scores.sum(0.0) / label_not_agreed_scores.size)**2 }.sum(0.0) / label_not_agreed_scores.size)}"
best_cutoff = 0
best_cutoff_score = 0
(0..100)
.step(1)
.each do |cutoff|
score =
label_agreed_scores.count { _1 > cutoff } + label_not_agreed_scores.count { _1 <= cutoff }
if score > best_cutoff_score
best_cutoff_score = score
best_cutoff = cutoff
end
end
puts "Recommended ai_toxicity_flag_threshold_#{label} value: #{best_cutoff}"
end
end