Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
69 changes: 69 additions & 0 deletions app/jobs/faultline_cleanup_job.rb
Original file line number Diff line number Diff line change
@@ -0,0 +1,69 @@
# Enforces Faultline's two retention settings, neither of which the gem
# enforces on its own (config/initializers/faultline.rb):
#
# * apm_retention_days -- request traces and the CPU profiles attached to
# them. The gem ships `rake faultline:apm:cleanup` for this; the job calls
# the same model methods.
# * retention_days -- error occurrences and their context rows. The gem only
# stores this number and leaves the deletion to "a cron job or Sidekiq
# scheduler", so it is done here.
#
# GoodJob's cron runs this nightly (config.good_job.cron in
# config/application.rb). Without it both sets of tables grow without bound.
class FaultlineCleanupJob < ApplicationJob
queue_as :default

def perform
traces = cleanup_apm
occurrences = cleanup_errors

Rails.logger.info(
"[Faultline] Cleanup removed #{traces} APM traces and #{occurrences} error occurrences"
)

{ traces: traces, occurrences: occurrences }
end

private

def cleanup_apm
return 0 unless Faultline::RequestTrace.table_exists?

Faultline::RequestProfile.cleanup! if Faultline::RequestProfile.table_exists?
Faultline::RequestTrace.cleanup!
end

# Occurrences are deleted with delete_all, which skips ActiveRecord
# callbacks, so the context rows (no ON DELETE CASCADE) are removed first
# and the groups' occurrences_count counter cache is recomputed afterwards.
# A group whose occurrences have all aged out is removed too, unless it was
# marked "ignored" -- keeping it is what stops a recurrence from opening a
# fresh, alerting group.
def cleanup_errors
retention_days = Faultline.configuration.retention_days
return 0 if retention_days.nil? # nil means keep forever

cutoff = retention_days.days.ago
stale = Faultline::ErrorOccurrence.where(created_at: ...cutoff)

Faultline::ErrorContext.where(error_occurrence_id: stale.select(:id)).delete_all
deleted = stale.delete_all
return 0 if deleted.zero?

# One statement recomputes every counter; there are no validations on
# ErrorGroup that a per-record save would add.
Faultline::ErrorGroup.update_all(<<~SQL.squish) # rubocop:disable Rails/SkipsModelValidations
occurrences_count = (
SELECT COUNT(*) FROM faultline_error_occurrences
WHERE faultline_error_occurrences.error_group_id = faultline_error_groups.id
)
SQL

Faultline::ErrorGroup.where(occurrences_count: 0)
.where(last_seen_at: ...cutoff)
.where.not(status: 'ignored')
.delete_all

deleted
end
end
5 changes: 5 additions & 0 deletions config/application.rb
Original file line number Diff line number Diff line change
Expand Up @@ -87,6 +87,11 @@ class Application < Rails::Application
class: 'PendingRequestsNotificationJob',
args: [ 'weekly' ],
description: 'Pending extension request digests for courses set to weekly'
},
faultline_cleanup: {
cron: '30 3 * * * America/Los_Angeles',
class: 'FaultlineCleanupJob',
description: 'Delete Faultline error data and APM traces past their retention windows'
}
}

Expand Down
91 changes: 61 additions & 30 deletions config/initializers/faultline.rb
Original file line number Diff line number Diff line change
Expand Up @@ -11,15 +11,23 @@
# Method to get current user in controllers
config.user_method = :current_user

# Custom context - add extra data to every error occurrence
# Receives request and Rack env, should return a hash
# config.custom_context = lambda { |request, env|
# controller = env["action_controller.instance"]
# {
# account_id: controller&.current_account&.id,
# tenant: request.subdomain
# }
# }
# Custom context - add extra data to every error occurrence.
#
# Faultline already records the `User` returned by `current_user` above
# (its id and email show up on the occurrence as the "user"), but that only
# works when the exception reaches the middleware with a controller instance
# in the Rack env. Errors raised earlier in the stack -- in another
# middleware, in the session store, before the controller is built -- have
# no controller to ask, and an anonymous request resolves to a NullUser whose
# id is nil. So we also attach the Canvas uid straight from the session as a
# context entry, which is the "basic user ID" the team looks up first: it is
# what the Canvas dashboard and the users table are keyed on, and it is
# present for every signed-in request regardless of where the error came
# from. Each key becomes a Faultline::ErrorContext row on the occurrence.
config.custom_context = lambda { |request, _env|
canvas_uid = request.session[:user_id]
{ canvas_uid: canvas_uid.presence }.compact
}

# =============================================================================
# Error Filtering
Expand Down Expand Up @@ -203,8 +211,12 @@
# This captures errors from background jobs and explicit Rails.error calls
config.register_error_subscriber = true

# Paths to ignore (no error tracking for these)
config.middleware_ignore_paths = ["/assets", "/up", "/health", "/admin/errors"]
# Paths to ignore (no error tracking for these).
# /status/health_check is the load balancer health check endpoint
# (StatusController#health_check); it reports database failures in its JSON
# body rather than raising. /admin/faultline is the engine itself
# (config/routes.rb).
config.middleware_ignore_paths = ["/assets", "/status/health_check", "/admin/faultline"]

# =============================================================================
# Data Configuration
Expand All @@ -213,20 +225,31 @@
# Maximum backtrace lines to store per occurrence
config.backtrace_lines_limit = 50

# How long to keep error data in days (nil = forever)
# Consider setting up a cleanup job if you have high error volume
# How long to keep error data in days (nil = forever). Faultline only stores
# this number; FaultlineCleanupJob (nightly via GoodJob's cron, see
# config.good_job.cron in config/application.rb) is what deletes occurrences
# older than this and the groups left empty by it.
config.retention_days = 90

# =============================================================================
# Callbacks (Advanced)
# =============================================================================

# Before tracking - return false to skip tracking this error
# config.before_track = lambda { |exception, context|
# # Example: Skip timeout errors
# return false if exception.message.include?("Timeout")
# true
# }
# Before tracking - return false to skip tracking this error.
#
# An unhandled request exception reaches Faultline twice. The Rack middleware
# (enable_middleware above, innermost in the stack) sees it first and records
# the request, the signed-in user and the captured locals. It then re-raises,
# and ActionDispatch::Executor at the top of the stack reports the very same
# exception to Rails.error with source "application.action_dispatch", which
# the error subscriber would turn into a second occurrence with no user and
# no URL -- doubling every count and alert threshold and leaving half the
# occurrences anonymous. Drop that second report. The subscriber still
# handles everything the middleware cannot see: background jobs (source
# "application.active_job" / "good_job") and explicit Rails.error calls.
config.before_track = lambda { |_exception, context|
context[:source] != "application.action_dispatch"
}

# After tracking - for custom integrations
# config.after_track = lambda { |error_group, occurrence|
Expand All @@ -249,17 +272,25 @@

# Enable basic APM to track request performance metrics.
# Captures response times, database queries, and throughput per endpoint.
# config.enable_apm = true

# Sample rate for high-traffic apps (0.0 to 1.0, default: 1.0 = every request)
# config.apm_sample_rate = 1.0

# Paths to ignore for APM (defaults to middleware_ignore_paths if nil)
# config.apm_ignore_paths = ["/assets", "/up", "/health", "/faultline"]

# How long to keep APM traces in days (default: 30)
# Use `rake faultline:apm:cleanup` to remove old traces.
# config.apm_retention_days = 30
# The dashboard lives at /admin/faultline/performance.
config.enable_apm = true

# Sample rate (0.0 to 1.0, 1.0 = every request). Each sampled request costs
# an extra INSERT (plus span JSON) after the response is sent, so we trace
# 30% of requests: enough to get meaningful p95s per endpoint without
# tripling the write load on a small database.
config.apm_sample_rate = 0.3

# Paths to ignore for APM (defaults to middleware_ignore_paths if nil).
# Faultline's own routes are always ignored; the load balancer polls
# /status/health_check constantly and would swamp the traces.
config.apm_ignore_paths = ["/assets", "/status/health_check", "/admin/faultline"]

# How long to keep APM traces (and their profiles) in days. Enforced by
# FaultlineCleanupJob, which GoodJob's cron runs nightly (see
# config.good_job.cron in config/application.rb). `rake faultline:apm:cleanup`
# does the APM half of that by hand.
config.apm_retention_days = 30

# --- Span Collection (Waterfall Visualization) ---
# Capture detailed spans for SQL, HTTP, Redis, and view rendering.
Expand Down
2 changes: 1 addition & 1 deletion config/routes.rb
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
Rails.application.routes.draw do
mount Faultline::Engine, at: "/admin/errors"
mount Faultline::Engine, at: "/admin/faultline"
if Rails.env.development?
mount LetterOpenerWeb::Engine, at: "/letter_opener"
end
Expand Down
9 changes: 9 additions & 0 deletions docs/developers.md
Original file line number Diff line number Diff line change
Expand Up @@ -130,6 +130,7 @@ occurrence is enqueued once even if several processes are running.
| `pending_digests_hourly` | Top of every hour | `PendingRequestsNotificationJob('hourly')` |
| `pending_digests_daily` | 4:00 PM PT daily | `PendingRequestsNotificationJob('daily')` |
| `pending_digests_weekly` | 4:00 PM PT Thursdays | `PendingRequestsNotificationJob('weekly')` |
| `faultline_cleanup` | 3:30 AM PT daily | `FaultlineCleanupJob` |

Each notification run emails the courses whose **Pending Request Notifications**
setting matches that frequency and that currently have pending requests.
Expand All @@ -142,6 +143,14 @@ access token, so syncs performed with different instructors' tokens do not
consume one shared quota; courses that share an instructor can still share that
token's quota. See the [Canvas API throttling documentation](https://developerdocs.instructure.com/services/canvas/basics/file.throttling).

The Faultline cleanup enforces the two retention windows set in
`config/initializers/faultline.rb`, which the gem records but never acts on:
error occurrences older than `retention_days` (and the groups they leave empty,
except ones marked ignored) and APM request traces older than
`apm_retention_days`. Faultline samples 30% of requests, so the traces table
would otherwise grow without bound. `bundle exec rake faultline:apm:cleanup`
does the APM half by hand.

Admins can inspect queues, schedules and past runs at `/admin/good_job`. To send a
digest by hand (locally, or to backfill after downtime):

Expand Down
67 changes: 67 additions & 0 deletions spec/config/faultline_spec.rb
Original file line number Diff line number Diff line change
Expand Up @@ -19,4 +19,71 @@
it 'subscribes to the Rails error reporter so job errors are tracked' do
expect(config.register_error_subscriber).to be true
end

describe 'dashboard' do
it 'is mounted at /admin/faultline' do
expect(Rails.application.routes.url_helpers.faultline_path).to eq('/admin/faultline')
end

it 'does not track its own requests or the load balancer health check' do
expect(config.middleware_ignore_paths).to include('/admin/faultline', '/status/health_check')
expect(config.middleware_ignore_paths).not_to include('/admin/errors')
end
end

describe 'user identification' do
it "records the controller's current_user on each occurrence" do
expect(config.user_class).to eq('User')
expect(config.user_method).to eq(:current_user)
end

it 'attaches the Canvas uid from the session as context' do
request = instance_double(ActionDispatch::Request, session: { user_id: '12345' })

expect(config.custom_context.call(request, {})).to eq(canvas_uid: '12345')
end

it 'adds no context for anonymous requests' do
request = instance_double(ActionDispatch::Request, session: {})

expect(config.custom_context.call(request, {})).to eq({})
end
end

describe 'duplicate suppression' do
let(:before_track) { config.before_track }

it "drops ActionDispatch's re-report of an exception the middleware already tracked" do
expect(before_track.call(StandardError.new, source: 'application.action_dispatch')).to be false
end

it 'keeps errors from jobs and explicit Rails.error calls' do
expect(before_track.call(StandardError.new, source: 'application.active_job')).to be true
expect(before_track.call(StandardError.new, source: 'good_job')).to be true
expect(before_track.call(StandardError.new, source: 'application')).to be true
end

it 'keeps errors tracked by the middleware itself' do
expect(before_track.call(StandardError.new, request: nil, user: nil)).to be true
end
end

describe 'APM' do
it 'is enabled' do
expect(config.enable_apm).to be true
end

it 'samples 30% of requests' do
expect(config.apm_sample_rate).to eq(0.3)
end

it 'skips the load balancer health check and its own dashboard' do
expect(config.resolved_apm_ignore_paths).to include('/status/health_check', '/admin/faultline')
end

it 'keeps traces for 30 days and errors for 90' do
expect(config.apm_retention_days).to eq(30)
expect(config.retention_days).to eq(90)
end
end
end
10 changes: 10 additions & 0 deletions spec/config/good_job_cron_spec.rb
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,16 @@
expect(occurrence.to_t.in_time_zone('America/Los_Angeles').strftime('%A %H:%M')).to eq('Thursday 16:00')
end

it 'prunes Faultline data once a night, after the enrollment sweep starts' do
entry = cron.fetch(:faultline_cleanup)
schedule = Fugit.parse_cron(entry[:cron])
occurrences = schedule.within(Time.utc(2026, 1, 1, 7, 59)..Time.utc(2026, 1, 2, 7, 59))

expect(entry[:class]).to eq('FaultlineCleanupJob')
expect(occurrences.size).to eq(1)
expect(occurrences.first.to_t.in_time_zone('America/Los_Angeles').strftime('%H:%M')).to eq('03:30')
end

it 'runs the enrollment sync sweep daily at 3:00 AM Pacific' do
schedule = Fugit.parse_cron(enrollment_sync_entry[:cron])
occurrences = schedule.within(Time.utc(2026, 1, 1, 7, 59)..Time.utc(2026, 1, 2, 7, 59))
Expand Down
Loading
Loading