Compare commits

...
Author SHA1 Message Date
eminuxandCursor 5a0ca8ef1c Corregge cloud-init montato, PublisherOnline multi-nodo e E2E locale-aware.
Senza volume stream-node i nodi Hetzner nascevano senza MediaMTX; sync live usa ora Client.for_session e rtmpconns (MediaMTX 1.20).

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-11 08:48:28 +02:00
eminuxandCursor 714602f3da Aggiunge overlay Compose per collaudo con RTMP su porta 11935.
Isola MediaMTX dal :1935 di produzione sullo stesso IP pubblico e documenta env/helper per il container Proxmox di test.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-10 20:29:47 +02:00
eminuxandCursor baac3a512b Preserva drain/offline su home e genera slate offline nel cloud-init.
Evita che ensure_home_from_env! annulli il drain a ogni allocate, e prepara /slates/offline.mp4 sul nodo Cloud per create_path MediaMTX.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-10 12:29:23 +02:00
eminuxandCursor 559284f0b2 Corregge il provisioning Hetzner: Faraday, cloud-init e default cpx12.
Sistemati path API /v1, AppArmor/auth MediaMTX sul nodo e defaults nbg1 così lo smoke Cloud è ripetibile.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-10 12:20:36 +02:00
eminuxandCursor e5fc925bae Completa il hardening autoscaler (Fase 4): budget, kill-switch e runbook.
Drain sicuro in scale-in, alert Ops su overflow e controlli admin senza abilitare il deploy in produzione.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-09 20:00:15 +02:00
eminuxandCursor 318a319608 Aggiunge l'autoscaler streaming con warm spare e kill-switch.
Scala i nodi overflow quando gli slot calano, mantiene una spare a caldo sotto carico e spegne gli idle, disattivabile con STREAM_AUTOSCALE_ENABLED.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-09 19:56:58 +02:00
eminuxandCursor 0b55d5a8fa Rende sticky e capacizzati i relay YouTube su coda dedicata.
Evita doppi ffmpeg tra host: stop solo sull'owner, ensure con requeue a capacità piena e coda Sidekiq youtube_relay isolata.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-09 19:54:04 +02:00
eminuxandCursor c3878cdc6d Aggiunge registry nodi stream e provisioning Hetzner/lab per lo scale-out.
Prepara l'architettura multi-nodo (MediaMTX+ffmpeg) con assignment URL per sessione, admin di provision/drain e provider Cloud/DNS astratti verso mltv-stream.net.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-08-09 19:52:47 +02:00
69 changed files with 3130 additions and 95 deletions
@@ -0,0 +1,67 @@
# frozen_string_literal: true
module Admin
class StreamNodesController < Admin::BaseController
def index
Streams::NodeRegistry.ensure_home_from_env!
@nodes = StreamNode.order(:role, :slug)
@dns_provider = ENV.fetch("STREAM_DNS_PROVIDER", "lab")
@cloud_provider = ENV.fetch("STREAM_CLOUD_PROVIDER", "local_lab")
@lab_hosts = lab_dns_snippet
@hetzner_configured = ENV["HCLOUD_TOKEN"].present?
@autoscale_metrics = Streams::Autoscaler.metrics
end
def create
kind = params[:kind].to_s
node =
if kind == "cloud"
Streams::NodeProvisioner.new.provision_cloud!
else
Streams::NodeProvisioner.new.provision_lab!
end
redirect_to admin_stream_nodes_path,
notice: t("admin.flash.stream_node_created", slug: node.slug)
rescue Streams::NodeProvisioner::Error, Streams::CloudProviders::Error, Streams::DnsProviders::Error,
KeyError => e
redirect_to admin_stream_nodes_path, alert: e.message
end
def drain
node = StreamNode.find(params[:id])
Streams::NodeProvisioner.new.drain!(node)
redirect_to admin_stream_nodes_path, notice: t("admin.flash.stream_node_draining", slug: node.slug)
rescue Streams::NodeProvisioner::Error => e
redirect_to admin_stream_nodes_path, alert: e.message
end
def destroy
node = StreamNode.find(params[:id])
Streams::NodeProvisioner.new.decommission!(node)
redirect_to admin_stream_nodes_path, notice: t("admin.flash.stream_node_destroyed", slug: node.slug)
rescue Streams::NodeProvisioner::BusyError, Streams::NodeProvisioner::Error,
Streams::CloudProviders::Error, Streams::DnsProviders::Error => e
redirect_to admin_stream_nodes_path, alert: e.message
end
def kill_switch
Streams::Autoscaler.engage_kill_switch!
redirect_to admin_stream_nodes_path, notice: t("admin.flash.autoscale_kill_on")
end
def clear_kill_switch
Streams::Autoscaler.clear_kill_switch!
redirect_to admin_stream_nodes_path, notice: t("admin.flash.autoscale_kill_off")
end
private
def lab_dns_snippet
return "" unless @dns_provider == "lab"
Streams::DnsProviders::Lab.new.hosts_file_snippet
rescue Redis::BaseError
""
end
end
end
@@ -180,6 +180,7 @@ module Api
platform: session.platform,
rtmp_ingest_url: session.rtmp_ingest_url,
hls_playback_url: session.hls_playback_url,
stream_node: session.stream_node&.slug,
watch_page_url: session.matchlivetv_platform? ? session.watch_page_url : nil,
share_url: session.share_url,
youtube_watch_url: session.youtube_watch_url,
@@ -6,7 +6,7 @@ class CleanupExpiredSessionsJob
.where("updated_at < ?", 6.hours.ago)
.find_each do |session|
session.fail! if session.may_fail?
Mediamtx::Client.new.delete_path(session)
Mediamtx::Client.for_session(session).delete_path(session)
end
end
end
@@ -0,0 +1,33 @@
# frozen_string_literal: true
module Streams
class AutoscalerJob
include Sidekiq::Job
sidekiq_options retry: 1, queue: "default"
INTERVAL_SECS = ENV.fetch("STREAM_AUTOSCALE_INTERVAL_SECS", "60").to_i
REDIS_CHAIN_KEY = "streams:autoscaler:chain"
def self.ensure_chain
return unless redis
return if redis.get(REDIS_CHAIN_KEY)
redis.set(REDIS_CHAIN_KEY, "1", ex: INTERVAL_SECS * 2)
perform_in(INTERVAL_SECS)
end
def self.redis
@redis ||= Redis.new(url: ENV.fetch("REDIS_URL", "redis://localhost:6379/0"))
rescue Redis::CannotConnectError
nil
end
def perform
Streams::Autoscaler.reconcile!
ensure
self.class.redis&.set(REDIS_CHAIN_KEY, "1", ex: INTERVAL_SECS * 2)
self.class.perform_in(INTERVAL_SECS)
end
end
end
+2 -2
View File
@@ -1,6 +1,6 @@
# Avvia/riavvia il relay YouTube solo nel container sidekiq (YOUTUBE_RELAY_WORKER=1).
# Avvia/riavvia il relay YouTube solo sui worker con YOUTUBE_RELAY_WORKER=1 (coda youtube_relay).
class YoutubeRelayEnsureJob < ApplicationJob
queue_as :default
queue_as Streams::YoutubeRelay::QUEUE
def perform(session_id)
session = StreamSession.find_by(id: session_id)
+15 -3
View File
@@ -1,10 +1,22 @@
# Ferma il relay solo sull'owner. Se il job gira su un altro host, requeue breve.
class YoutubeRelayStopJob < ApplicationJob
queue_as :default
queue_as Streams::YoutubeRelay::QUEUE
def perform(session_id)
discard_on ActiveJob::DeserializationError
def perform(session_id, attempts = 0)
session = StreamSession.find_by(id: session_id)
return unless session
Streams::YoutubeRelay.stop_on_worker!(session) if Streams::YoutubeRelay.worker?
unless Streams::YoutubeRelay.worker?
# Solo i worker relay processano questa coda in modo utile.
return
end
result = Streams::YoutubeRelay.stop_on_worker!(session)
return unless result == :wrong_host
return if attempts >= 30
self.class.set(wait: 2.seconds).perform_later(session_id, attempts + 1)
end
end
+1 -1
View File
@@ -4,7 +4,7 @@ module Ops
KINDS = %w[
disk_space recordings_size service_down http_public http_rails rails_latency
sidekiq_stale sidekiq_dead log_pattern garage_storage
sidekiq_stale sidekiq_dead log_pattern garage_storage stream_overflow
].freeze
SEVERITIES = %w[critical warning info].freeze
STATUSES = %w[open acknowledged resolved].freeze
+38
View File
@@ -0,0 +1,38 @@
# frozen_string_literal: true
# Nodo streaming (MediaMTX [+ relay ffmpeg]). Fase 0: registry + assignment URL.
class StreamNode < ApplicationRecord
ROLES = %w[home cloud lab].freeze
STATUSES = %w[provisioning ready draining offline error].freeze
PROVIDERS = %w[local proxmox_lab hetzner].freeze
has_many :stream_sessions, dependent: :nullify
validates :slug, presence: true, uniqueness: true
validates :hostname, presence: true
validates :role, inclusion: { in: ROLES }
validates :status, inclusion: { in: STATUSES }
validates :provider, inclusion: { in: PROVIDERS }
validates :rtmp_base_url, :hls_base_url, :api_base_url, presence: true
validates :max_publishers, :max_relays, numericality: { greater_than: 0 }
scope :ready, -> { where(status: "ready") }
scope :allocatable, -> { ready }
# Sessioni che occupano uno slot path MediaMTX su questo nodo.
def occupying_sessions
stream_sessions.where(status: %w[idle connecting live reconnecting paused])
end
def active_publishers
occupying_sessions.count
end
def free_slots
[max_publishers - active_publishers, 0].max
end
def allocatable?
status == "ready" && free_slots.positive?
end
end
+25 -3
View File
@@ -7,6 +7,7 @@ class StreamSession < ApplicationRecord
belongs_to :match
belongs_to :user
belongs_to :stream_node, optional: true
has_many :stream_events, dependent: :destroy
has_one :score_state, dependent: :destroy
has_many :device_states, dependent: :destroy
@@ -74,7 +75,7 @@ class StreamSession < ApplicationRecord
def rtmp_ingest_url
# RootEncoder richiede rtmp://host:port/app/stream (due segmenti).
# MediaMTX path = live/match_{uuid} (no ?token= nel path).
"#{MatchLiveTv.mediamtx_rtmp_url}/#{mediamtx_path_name}"
"#{rtmp_base_url.chomp('/')}/#{mediamtx_path_name}"
end
def mediamtx_path_name
@@ -90,8 +91,29 @@ class StreamSession < ApplicationRecord
end
def hls_playback_url
base = MatchLiveTv.hls_public_url.chomp("/")
"#{base}/#{effective_hls_path_name}/index.m3u8"
"#{hls_base_url.chomp('/')}/#{effective_hls_path_name}/index.m3u8"
end
def rtmp_base_url
stream_node&.rtmp_base_url.presence || MatchLiveTv.mediamtx_rtmp_url
end
def hls_base_url
stream_node&.hls_base_url.presence || MatchLiveTv.hls_public_url
end
def mediamtx_api_base_url
stream_node&.api_base_url.presence || MatchLiveTv.mediamtx_api_url
end
def mediamtx_internal_rtmp_url
stream_node&.internal_rtmp_url.presence ||
ENV.fetch("MEDIAMTX_INTERNAL_RTMP_URL", "rtmp://mediamtx:1935")
end
def mediamtx_internal_hls_url
stream_node&.internal_hls_url.presence ||
ENV.fetch("MEDIAMTX_HLS_URL", "http://mediamtx:8888")
end
def effective_hls_path_name
+17
View File
@@ -4,7 +4,12 @@ module Mediamtx
class Client
class Error < StandardError; end
def self.for_session(session)
new(base_url: session.mediamtx_api_base_url)
end
def initialize(base_url: MatchLiveTv.mediamtx_api_url)
@base_url = base_url
@conn = Faraday.new(url: base_url) do |f|
f.request :json
f.response :json
@@ -12,6 +17,8 @@ module Mediamtx
end
end
attr_reader :base_url
def create_path(session)
path = session.mediamtx_path_name
# record: false finché non c'è publisher — con alwaysAvailable MediaMTX registrerebbe
@@ -98,6 +105,16 @@ module Mediamtx
[]
end
def list_rtmp_conns
response = @conn.get("/v3/rtmpconns/list")
return [] unless response.success?
body = response.body
body.is_a?(Hash) ? (body["items"] || []) : []
rescue Error, Faraday::Error
[]
end
def online_path_names
Set.new(list_paths.filter_map { |item| item["name"] if item["online"] })
end
@@ -4,17 +4,37 @@ module Mediamtx
module_function
def active?(session)
return true if rtmp_publisher?(session)
active_path?(path_info(session))
end
def path_info(session)
Client.new.list_paths.find { |i| i["name"] == session.mediamtx_path_name }
Client.for_session(session).list_paths.find { |i| i["name"] == session.mediamtx_path_name }
end
# MediaMTX <=1.19: online + source.type=rtmpConn.
# MediaMTX 1.20+: online/source spesso null anche con publisher; usare rtmpconns.
def active_path?(info)
return false unless info
return true if info["online"] == true && rtmp_source?(info.dig("source", "type"))
info["online"] == true && info.dig("source", "type") == "rtmpConn"
false
end
def rtmp_publisher?(session)
path = session.mediamtx_path_name.to_s
return false if path.blank?
Client.for_session(session).list_rtmp_conns.any? do |conn|
conn_path = conn["path"].to_s.sub(%r{\A/}, "")
next false unless conn_path == path
state = conn["state"].to_s
state.empty? || state == "publish" || state == "idle"
end
rescue StandardError
false
end
def h264_video?(info)
@@ -26,12 +46,14 @@ module Mediamtx
end
def video_publishing?(session)
info = path_info(session)
return false unless active_path?(info)
# Slate alwaysAvailable ha H264 ma non è il telefono.
return false unless info.dig("source", "type") == "rtmpConn"
return false unless active?(session)
info = path_info(session)
h264_video?(info)
end
def rtmp_source?(type)
type.to_s.match?(/\Artmps?Conn\z/)
end
end
end
@@ -12,7 +12,7 @@ module Mediamtx
return @session if @session.terminal?
path_info = Mediamtx::PublisherOnline.path_info(@session)
publisher_online = Mediamtx::PublisherOnline.active_path?(path_info)
publisher_online = Mediamtx::PublisherOnline.active?(@session)
if publisher_online
clear_publisher_misses!(@session.id)
@@ -86,7 +86,7 @@ module Mediamtx
key = format("youtube:slate_disabled:%s", session.id)
return unless redis.set(key, "1", nx: true, ex: 48.hours.to_i)
Client.new.set_always_available(session, enabled: false)
Client.for_session(session).set_always_available(session, enabled: false)
rescue Client::Error => e
redis.del(format("youtube:slate_disabled:%s", session.id))
Rails.logger.warn("[PublisherSync] disable slate session=#{session.id}: #{e.message}")
@@ -95,7 +95,7 @@ module Mediamtx
def restore_slate_path!(session)
return if session.platform == "matchlivetv"
Client.new.set_always_available(session, enabled: true)
Client.for_session(session).set_always_available(session, enabled: true)
rescue Client::Error => e
Rails.logger.warn("[PublisherSync] enable slate session=#{session.id}: #{e.message}")
end
@@ -128,7 +128,7 @@ module Mediamtx
return
end
Client.new.set_path_recording(session, enabled: enabled)
Client.for_session(session).set_path_recording(session, enabled: enabled)
redis.set(key, desired, ex: 48.hours.to_i)
mark_recording_patch!(session.id)
rescue Client::Error => e
@@ -183,7 +183,7 @@ module Mediamtx
key = format("youtube_relay:sched:%s", session.id)
return unless redis.set(key, "1", nx: true, ex: 10)
YoutubeRelayEnsureJob.perform_later(session.id)
YoutubeRelayEnsureJob.set(queue: Streams::YoutubeRelay::QUEUE).perform_later(session.id)
end
def redis
+54 -1
View File
@@ -44,7 +44,8 @@ module Ops
check_sidekiq_heartbeat,
check_sidekiq_dead,
check_http_rails,
check_rails_latency
check_rails_latency,
check_stream_overflow
]
findings << check_http_public if public_check_due?
findings
@@ -161,6 +162,58 @@ module Ops
fail_finding("garage_storage", "warning", "garage_storage:head", "Garage storage non raggiungibile", e.message)
end
def check_stream_overflow
return ok_finding("stream_overflow:skip", "Nodi stream non migrati") unless ActiveRecord::Base.connection.data_source_exists?("stream_nodes")
orphan_hours = ENV.fetch("STREAM_OVERFLOW_ORPHAN_HOURS", "3").to_i
orphans = StreamNode.where.not(slug: "home").where(status: %w[ready draining]).select do |n|
n.active_publishers.zero? && n.created_at < orphan_hours.hours.ago
end
metrics = Streams::Autoscaler.metrics
over_budget = !metrics[:within_budget]
at_max = metrics[:overflow_nodes] >= metrics[:max_overflow_nodes] && metrics[:free_slots] <= metrics[:soft_free_slots]
if orphans.any?
return Finding.new(
kind: "stream_overflow",
severity: "warning",
healthy: false,
title: "Nodi stream overflow idle",
message: "#{orphans.size} nodo/i idle da >#{orphan_hours}h: #{orphans.map(&:slug).join(', ')}",
metadata: { "slugs" => orphans.map(&:slug) },
fingerprint: "stream_overflow:orphan_idle"
)
end
if over_budget
return Finding.new(
kind: "stream_overflow",
severity: "warning",
healthy: false,
title: "Budget overflow streaming",
message: "Stima €#{metrics[:estimated_monthly_eur]}/mese > budget €#{metrics[:monthly_budget_eur]}",
metadata: metrics.transform_keys(&:to_s),
fingerprint: "stream_overflow:budget"
)
end
if at_max
return Finding.new(
kind: "stream_overflow",
severity: "warning",
healthy: false,
title: "Capacità stream al massimo",
message: "overflow=#{metrics[:overflow_nodes]}/#{metrics[:max_overflow_nodes]} free_slots=#{metrics[:free_slots]}",
metadata: metrics.transform_keys(&:to_s),
fingerprint: "stream_overflow:at_max"
)
end
ok_finding("stream_overflow:ok", "Overflow streaming OK")
rescue StandardError => e
fail_finding("stream_overflow", "warning", "stream_overflow:error", "Check overflow fallito", e.message)
end
def check_sidekiq_heartbeat
redis = Redis.new(url: ENV.fetch("REDIS_URL", "redis://localhost:6379/0"))
last = redis.get(Ops::HealthMonitorJob::HEARTBEAT_KEY).to_i
@@ -95,7 +95,7 @@ module Recordings
def cleanup_mediamtx_path(session)
return unless session.status.in?(%w[ended error])
Mediamtx::Client.new.delete_path(session)
Mediamtx::Client.for_session(session).delete_path(session)
rescue Mediamtx::Client::Error => e
@logger.warn("[Recordings::CleanupLocal] delete_path #{session.id}: #{e.message}")
end
@@ -95,7 +95,7 @@ module Recordings
def cleanup_mediamtx_path
Mediamtx::PublisherSync.forget_recording_state!(@session.id)
Mediamtx::Client.new.delete_path(@session)
Mediamtx::Client.for_session(@session).delete_path(@session)
rescue Mediamtx::Client::Error => e
Rails.logger.warn("[Recordings::UploadFromSession] delete_path: #{e.message}")
end
+13 -3
View File
@@ -34,11 +34,19 @@ module Sessions
end
StreamSession.transaction do
session.stream_node = Streams::NodeRegistry.allocate!
session.save!
Scoring::Engine.ensure_score_for(session)
mtx = Mediamtx::Client.new
mtx.create_path(session)
log_event(session, "pairing", { created: true, platform: session.platform })
Mediamtx::Client.for_session(session).create_path(session)
log_event(
session,
"pairing",
{
created: true,
platform: session.platform,
stream_node: session.stream_node&.slug
}
)
end
if session.platform == "youtube"
@@ -46,6 +54,8 @@ module Sessions
end
session
rescue Streams::NodeRegistry::NoCapacityError => e
raise Teams::EntitlementError.new(e.message, code: "stream_capacity_exhausted")
end
private
+2 -2
View File
@@ -9,11 +9,11 @@ module Sessions
@session.pause! if @session.may_pause?
Mediamtx::PublisherSync.forget_recording_state!(@session.id)
begin
Mediamtx::Client.new.set_path_recording(@session, enabled: false)
Mediamtx::Client.for_session(@session).set_path_recording(@session, enabled: false)
rescue Mediamtx::Client::Error => e
Rails.logger.warn("[Sessions::Pause] disable recording: #{e.message}")
end
Mediamtx::Client.new.set_always_available(@session, enabled: true)
Mediamtx::Client.for_session(@session).set_always_available(@session, enabled: true)
# Slate su path per HLS in pausa; RTMP telefono si ferma via comando app.
log_event("paused")
SessionChannel.broadcast_message(@session, { type: "command", action: "pause_stream" })
+1 -1
View File
@@ -33,7 +33,7 @@ module Sessions
end
def remove_mediamtx_paths!
Mediamtx::Client.new.delete_path(@session)
Mediamtx::Client.for_session(@session).delete_path(@session)
rescue Mediamtx::Client::Error => e
Rails.logger.warn("[Sessions::Stop] delete_path #{@session.id}: #{e.message}")
end
+236
View File
@@ -0,0 +1,236 @@
# frozen_string_literal: true
module Streams
# Scale-out / warm spare / scale-in dei nodi overflow (lab o Hetzner).
# Kill-switch: STREAM_AUTOSCALE_ENABLED!=1 OPPURE Redis streams:autoscaler:kill_switch=1.
class Autoscaler
Result = Struct.new(:actions, :metrics, :skipped, :error, keyword_init: true)
LOCK_KEY = "streams:autoscaler:lock"
KILL_SWITCH_KEY = "streams:autoscaler:kill_switch"
class << self
def enabled?
return false if kill_switch_engaged?
return false unless ENV["STREAM_AUTOSCALE_ENABLED"] == "1"
true
end
def kill_switch_engaged?
redis_get(KILL_SWITCH_KEY) == "1"
end
def engage_kill_switch!
redis_set(KILL_SWITCH_KEY, "1")
end
def clear_kill_switch!
redis_del(KILL_SWITCH_KEY)
end
def soft_free_slots
ENV.fetch("STREAM_AUTOSCALE_SOFT_FREE_SLOTS", "2").to_i
end
def warm_spare_min
ENV.fetch("STREAM_AUTOSCALE_WARM_SPARE", "1").to_i
end
def idle_minutes
ENV.fetch("STREAM_AUTOSCALE_IDLE_MINUTES", "30").to_i
end
def max_overflow_nodes
ENV.fetch("STREAM_AUTOSCALE_MAX_NODES", "5").to_i
end
def kind
ENV.fetch("STREAM_AUTOSCALE_KIND", "lab") # lab|cloud
end
def allow_cloud?
ENV["STREAM_AUTOSCALE_ALLOW_CLOUD"] == "1" && ENV["HCLOUD_TOKEN"].present?
end
def node_eur_per_hour
ENV.fetch("STREAM_AUTOSCALE_NODE_EUR_PER_HOUR", "0.015").to_f
end
def monthly_budget_eur
ENV.fetch("STREAM_AUTOSCALE_MONTHLY_BUDGET_EUR", "40").to_f
end
def estimated_monthly_eur(overflow_count = nil)
count = overflow_count || metrics[:overflow_nodes]
# Worst case: nodi sempre accesi 24/7
(count * node_eur_per_hour * 24 * 30).round(2)
end
def within_budget?(overflow_count = nil)
estimated_monthly_eur(overflow_count) <= monthly_budget_eur
end
def reconcile!(provisioner: nil)
return Result.new(skipped: true, actions: [], metrics: metrics) unless enabled?
redis = Redis.new(url: ENV.fetch("REDIS_URL", "redis://localhost:6379/0"))
unless redis.set(LOCK_KEY, worker_id, nx: true, ex: 55)
return Result.new(skipped: true, actions: [], metrics: metrics, error: "locked")
end
begin
new(provisioner: provisioner).reconcile!
ensure
redis.del(LOCK_KEY)
end
end
def metrics
NodeRegistry.ensure_home_from_env!
nodes = StreamNode.ready.to_a
overflow = StreamNode.where.not(slug: NodeRegistry::HOME_SLUG)
.where(status: %w[ready draining provisioning]).to_a
{
free_slots: nodes.sum(&:free_slots),
ready_nodes: nodes.size,
spare_ready: nodes.count { |n| n.slug != NodeRegistry::HOME_SLUG && n.active_publishers.zero? },
overflow_nodes: overflow.size,
soft_free_slots: soft_free_slots,
warm_spare_min: warm_spare_min,
max_overflow_nodes: max_overflow_nodes,
enabled: enabled?,
env_enabled: ENV["STREAM_AUTOSCALE_ENABLED"] == "1",
kill_switch: kill_switch_engaged?,
kind: kind,
allow_cloud: allow_cloud?,
estimated_monthly_eur: estimated_monthly_eur(overflow.size),
monthly_budget_eur: monthly_budget_eur,
within_budget: within_budget?(overflow.size)
}
end
def worker_id
ENV.fetch("HOSTNAME", "autoscaler")
end
def redis_get(key)
Redis.new(url: ENV.fetch("REDIS_URL", "redis://localhost:6379/0")).get(key)
rescue Redis::BaseError
nil
end
def redis_set(key, value)
Redis.new(url: ENV.fetch("REDIS_URL", "redis://localhost:6379/0")).set(key, value)
end
def redis_del(key)
Redis.new(url: ENV.fetch("REDIS_URL", "redis://localhost:6379/0")).del(key)
end
end
def initialize(provisioner: nil)
@provisioner = provisioner || NodeProvisioner.new
end
def reconcile!
actions = []
m = self.class.metrics
if need_capacity?(m) && can_provision?(m)
provision_overflow!
actions << :scale_out
m = self.class.metrics
elsif need_capacity?(m) && !can_provision?(m)
actions << :blocked_capacity
Rails.logger.warn("[Streams::Autoscaler] capacity needed but blocked metrics=#{m.inspect}")
end
if warm_spare_desired?(m) && m[:spare_ready] < self.class.warm_spare_min && can_provision?(m)
provision_overflow!
actions << :warm_spare
m = self.class.metrics
end
scale_in_candidates.each do |node|
next if keep_as_warm_spare?(node)
safe_scale_in!(node)
actions << :"scale_in_#{node.slug}"
m = self.class.metrics
rescue NodeProvisioner::BusyError, NodeProvisioner::Error => e
Rails.logger.warn("[Streams::Autoscaler] scale-in #{node.slug}: #{e.message}")
end
Rails.logger.info("[Streams::Autoscaler] actions=#{actions.inspect} metrics=#{m.inspect}")
Result.new(actions: actions, metrics: m, skipped: false)
end
private
def need_capacity?(m)
m[:free_slots] <= self.class.soft_free_slots
end
def warm_spare_desired?(m)
return false if self.class.warm_spare_min <= 0
need_capacity?(m) || overflow_in_use?
end
def overflow_in_use?
StreamNode.ready.where.not(slug: NodeRegistry::HOME_SLUG).any? { |n| n.active_publishers.positive? }
end
def can_provision?(m)
return false if m[:overflow_nodes] >= self.class.max_overflow_nodes
return false unless self.class.within_budget?(m[:overflow_nodes] + 1)
return false if self.class.kind == "cloud" && !self.class.allow_cloud?
true
end
def provision_overflow!
case self.class.kind
when "cloud"
raise NodeProvisioner::Error, "Cloud autoscale disabilitato (STREAM_AUTOSCALE_ALLOW_CLOUD / HCLOUD_TOKEN)" unless self.class.allow_cloud?
@provisioner.provision_cloud!
else
@provisioner.provision_lab!
end
end
def safe_scale_in!(node)
@provisioner.drain!(node) unless node.status == "draining"
node.reload
raise NodeProvisioner::BusyError, "sessioni ancora attive" if node.occupying_sessions.exists?
raise NodeProvisioner::Error, "idle insufficiente" unless idle_long_enough?(node)
@provisioner.decommission!(node)
end
def scale_in_candidates
StreamNode.where.not(slug: NodeRegistry::HOME_SLUG)
.where(status: %w[ready draining])
.order(:created_at)
.select { |n| n.active_publishers.zero? && idle_long_enough?(n) }
end
def idle_long_enough?(node)
idle_since(node) <= self.class.idle_minutes.minutes.ago
end
def idle_since(node)
last_end = node.stream_sessions.where(status: %w[ended error]).maximum(:ended_at)
last_end || node.created_at
end
def keep_as_warm_spare?(node)
return false unless warm_spare_desired?(self.class.metrics)
spares = StreamNode.ready.where.not(slug: NodeRegistry::HOME_SLUG).select { |n| n.active_publishers.zero? }
spares.size <= self.class.warm_spare_min && spares.map(&:id).include?(node.id)
end
end
end
@@ -0,0 +1,15 @@
# frozen_string_literal: true
module Streams
module CloudProviders
def self.build(name = ENV.fetch("STREAM_CLOUD_PROVIDER", "local_lab"))
case name.to_s
when "local_lab" then LocalLab.new
when "proxmox_lab" then ProxmoxLab.new
when "hetzner" then Hetzner.new
else
raise Error, "STREAM_CLOUD_PROVIDER sconosciuto: #{name}"
end
end
end
end
@@ -0,0 +1,35 @@
# frozen_string_literal: true
module Streams
module CloudProviders
class Error < StandardError; end
# Descrittore restituito da create_node / list.
Instance = Struct.new(
:id, :name, :public_ip, :private_ip, :status, :raw,
keyword_init: true
)
class Base
def create_node(name:, labels: {})
raise NotImplementedError
end
def destroy_node(instance_id)
raise NotImplementedError
end
def list_nodes(labels: {})
raise NotImplementedError
end
def wait_until_running(instance_id, timeout: 120)
raise NotImplementedError
end
def public_ip(instance_id)
raise NotImplementedError
end
end
end
end
@@ -0,0 +1,191 @@
# frozen_string_literal: true
require "faraday"
module Streams
module CloudProviders
# Hetzner Cloud — create/destroy server per nodi stream.
#
# ENV:
# HCLOUD_TOKEN (obbligatorio)
# HCLOUD_LOCATION (default fsn1)
# HCLOUD_SERVER_TYPE (default cpx21)
# HCLOUD_IMAGE (default debian-12)
# HCLOUD_SSH_KEY (nome chiave, default matchlivetv-stream)
# HCLOUD_NETWORK_ID (opzionale, private network / WireGuard prep)
# HCLOUD_USER_DATA_FILE (opzionale, cloud-init path)
class Hetzner < Base
# Trailing slash obbligatorio: path assoluti tipo "/servers" altrimenti droppano /v1.
API = "https://api.hetzner.cloud/v1/"
def initialize(token: ENV.fetch("HCLOUD_TOKEN"), conn: nil)
@token = token
@conn = conn
end
def create_node(name:, labels: {})
body = {
name: name,
server_type: ENV.fetch("HCLOUD_SERVER_TYPE", "cpx12"),
image: ENV.fetch("HCLOUD_IMAGE", "debian-12"),
location: ENV.fetch("HCLOUD_LOCATION", "nbg1"),
start_after_create: true,
labels: default_labels.merge(stringify_labels(labels)),
ssh_keys: [ENV.fetch("HCLOUD_SSH_KEY", "matchlivetv-stream-hetzner")],
public_net: {
enable_ipv4: true,
enable_ipv6: false
}
}
network_id = ENV["HCLOUD_NETWORK_ID"].presence
body[:networks] = [network_id.to_i] if network_id
user_data = cloud_init_user_data
body[:user_data] = user_data
data = post("servers", body)
server = data["server"] || {}
action = data["action"]
wait_action!(action) if action
instance = wait_until_running(server["id"].to_s)
instance.name = name
instance
end
def destroy_node(instance_id)
delete("servers/#{instance_id}")
true
end
def list_nodes(labels: {})
params = {}
label_selector = labels.map { |k, v| "#{k}=#{v}" }.join(",")
params[:label_selector] = label_selector if label_selector.present?
params[:label_selector] ||= "matchlivetv=true,role=stream-node"
data = get("servers", params)
Array(data["servers"]).map { |s| instance_from_server(s) }
end
def wait_until_running(instance_id, timeout: 180)
deadline = Time.now + timeout
loop do
data = get("servers/#{instance_id}")
server = data["server"]
status = server["status"]
if status == "running"
return instance_from_server(server)
end
raise Error, "Timeout attesa server Hetzner #{instance_id} (status=#{status})" if Time.now >= deadline
sleep 3
end
end
def public_ip(instance_id)
wait_until_running(instance_id).public_ip
end
private
def default_labels
{
"matchlivetv" => "true",
"role" => "stream-node",
"env" => ENV.fetch("STREAM_NODE_ENV", "prod")
}
end
def stringify_labels(labels)
labels.to_h.transform_keys(&:to_s).transform_values(&:to_s)
end
def instance_from_server(server)
public_ip = server.dig("public_net", "ipv4", "ip")
private_ip = Array(server["private_net"]).first&.dig("ip")
Instance.new(
id: server["id"].to_s,
name: server["name"],
public_ip: public_ip,
private_ip: private_ip.presence || public_ip,
status: server["status"],
raw: server
)
end
def cloud_init_user_data
path = ENV["HCLOUD_USER_DATA_FILE"].presence
if path.present?
raise Error, "HCLOUD_USER_DATA_FILE non leggibile nel container: #{path}" unless File.file?(path)
return File.read(path)
end
inline = ENV["HCLOUD_USER_DATA"].presence
raise Error, "Manca cloud-init: imposta HCLOUD_USER_DATA_FILE (montato) o HCLOUD_USER_DATA" if inline.blank?
inline
end
def conn
@conn ||= Faraday.new(url: API) do |f|
f.request :json
f.response :json, content_type: /\bjson$/
f.adapter Faraday.default_adapter
end
end
def auth_headers
{ "Authorization" => "Bearer #{@token}" }
end
def get(path, params = {})
response = conn.get(path) do |req|
req.headers.update(auth_headers)
req.params.update(params)
end
unwrap!(response)
end
def post(path, body)
response = conn.post(path) do |req|
req.headers.update(auth_headers)
req.body = body
end
unwrap!(response)
end
def delete(path)
response = conn.delete(path) do |req|
req.headers.update(auth_headers)
end
return {} if response.status == 204
unwrap!(response)
end
def unwrap!(response)
unless response.success?
raise Error, "Hetzner Cloud API #{response.status}: #{response.body.inspect}"
end
response.body.is_a?(Hash) ? response.body : {}
end
def wait_action!(action, timeout: 120)
return unless action.is_a?(Hash) && action["id"]
deadline = Time.now + timeout
id = action["id"]
loop do
data = get("actions/#{id}")
status = data.dig("action", "status")
return if status == "success"
raise Error, "Hetzner action #{id} failed: #{data.inspect}" if status == "error"
raise Error, "Timeout action Hetzner #{id}" if Time.now >= deadline
sleep 2
end
end
end
end
end
@@ -0,0 +1,45 @@
# frozen_string_literal: true
module Streams
module CloudProviders
# Lab senza API Proxmox: simula create/destroy e riusa MediaMTX home per i path.
# Utile per testare registry, assignment e admin senza secondi host.
class LocalLab < Base
def create_node(name:, labels: {})
Instance.new(
id: "sim-#{name}",
name: name,
public_ip: labels[:public_ip].presence || "127.0.0.1",
private_ip: labels[:private_ip].presence || "127.0.0.1",
status: "running",
raw: { simulated: true, labels: labels }
)
end
def destroy_node(instance_id)
true
end
def list_nodes(labels: {})
StreamNode.where(provider: "local", role: "lab").map do |node|
Instance.new(
id: node.provider_instance_id,
name: node.slug,
public_ip: node.metadata["public_ip"],
private_ip: node.metadata["private_ip"],
status: node.status == "ready" ? "running" : node.status,
raw: node.metadata
)
end
end
def wait_until_running(instance_id, timeout: 120)
Instance.new(id: instance_id, name: instance_id, status: "running")
end
def public_ip(instance_id)
"127.0.0.1"
end
end
end
end
@@ -0,0 +1,162 @@
# frozen_string_literal: true
require "faraday"
module Streams
module CloudProviders
# Clone/start/stop di VM template su Proxmox VE (API token).
#
# ENV richiesti:
# PROXMOX_API_URL, PROXMOX_TOKEN_ID, PROXMOX_TOKEN_SECRET,
# PROXMOX_NODE, PROXMOX_TEMPLATE_VMID
class ProxmoxLab < Base
def initialize(
api_url: ENV.fetch("PROXMOX_API_URL"),
token_id: ENV.fetch("PROXMOX_TOKEN_ID"),
token_secret: ENV.fetch("PROXMOX_TOKEN_SECRET"),
node: ENV.fetch("PROXMOX_NODE"),
template_vmid: ENV.fetch("PROXMOX_TEMPLATE_VMID"),
verify_ssl: ENV.fetch("PROXMOX_VERIFY_SSL", "false") == "true"
)
@api_url = api_url.to_s.chomp("/")
@token_id = token_id
@token_secret = token_secret
@node = node
@template_vmid = template_vmid.to_i
@verify_ssl = verify_ssl
end
def create_node(name:, labels: {})
newid = next_vmid
post("/nodes/#{@node}/qemu/#{@template_vmid}/clone", {
newid: newid,
name: name,
full: 1,
target: @node
})
post("/nodes/#{@node}/qemu/#{newid}/status/start", {})
wait_until_running(newid.to_s)
ip = public_ip(newid.to_s)
Instance.new(
id: newid.to_s,
name: name,
public_ip: ip,
private_ip: ip,
status: "running",
raw: { node: @node, vmid: newid, labels: labels }
)
end
def destroy_node(instance_id)
vmid = instance_id.to_i
begin
post("/nodes/#{@node}/qemu/#{vmid}/status/stop", { timeout: 30 })
rescue Error
# già spenta
end
sleep 2
delete("/nodes/#{@node}/qemu/#{vmid}", { purge: 1 })
true
end
def list_nodes(labels: {})
items = get("/nodes/#{@node}/qemu")
Array(items).filter_map do |row|
name = row["name"].to_s
next unless name.start_with?("mltv-stream-") || name.start_with?("ingest-")
Instance.new(
id: row["vmid"].to_s,
name: name,
public_ip: nil,
private_ip: nil,
status: row["status"],
raw: row
)
end
end
def wait_until_running(instance_id, timeout: 180)
deadline = Time.now + timeout
loop do
status = get("/nodes/#{@node}/qemu/#{instance_id}/status/current")
return Instance.new(id: instance_id.to_s, status: "running", raw: status) if status["status"] == "running"
raise Error, "Timeout attesa VM #{instance_id}" if Time.now >= deadline
sleep 3
end
end
def public_ip(instance_id)
agent = get("/nodes/#{@node}/qemu/#{instance_id}/agent/network-get-interfaces")
interfaces = agent.is_a?(Hash) ? agent["result"] : nil
Array(interfaces).each do |iface|
Array(iface["ip-addresses"]).each do |addr|
ip = addr["ip-address"].to_s
next if ip.blank? || ip.start_with?("127.") || ip.include?(":")
return ip
end
end
ENV["STREAM_LAB_FALLBACK_IP"].presence || "127.0.0.1"
rescue Error
ENV["STREAM_LAB_FALLBACK_IP"].presence || "127.0.0.1"
end
private
def next_vmid
used = Array(get("/cluster/resources", type: "vm")).map { |r| r["vmid"].to_i }
candidate = ENV.fetch("PROXMOX_VMID_START", "9100").to_i
candidate += 1 while used.include?(candidate)
candidate
end
def conn
@conn ||= Faraday.new(url: "#{@api_url}/api2/json") do |f|
f.request :url_encoded
f.response :json, content_type: /\bjson$/
f.adapter Faraday.default_adapter
f.ssl[:verify] = @verify_ssl
end
end
def auth_headers
{ "Authorization" => "PVEAPIToken=#{@token_id}=#{@token_secret}" }
end
def get(path, params = {})
response = conn.get(path) do |req|
req.headers.update(auth_headers)
req.params.update(params)
end
unwrap!(response)
end
def post(path, body = {})
response = conn.post(path) do |req|
req.headers.update(auth_headers)
req.body = body
end
unwrap!(response)
end
def delete(path, params = {})
response = conn.delete(path) do |req|
req.headers.update(auth_headers)
req.params.update(params)
end
unwrap!(response)
end
def unwrap!(response)
unless response.success?
raise Error, "Proxmox API #{response.status}: #{response.body.inspect}"
end
body = response.body
body.is_a?(Hash) && body.key?("data") ? body["data"] : body
end
end
end
end
@@ -0,0 +1,14 @@
# frozen_string_literal: true
module Streams
module DnsProviders
def self.build(name = ENV.fetch("STREAM_DNS_PROVIDER", "lab"))
case name.to_s
when "lab" then Lab.new
when "hetzner" then Hetzner.new
else
raise Error, "STREAM_DNS_PROVIDER sconosciuto: #{name}"
end
end
end
end
@@ -0,0 +1,21 @@
# frozen_string_literal: true
module Streams
module DnsProviders
class Error < StandardError; end
class Base
def upsert_a(name, ip)
raise NotImplementedError
end
def delete_a(name)
raise NotImplementedError
end
def resolve(name)
raise NotImplementedError
end
end
end
end
@@ -0,0 +1,106 @@
# frozen_string_literal: true
require "faraday"
require "cgi"
module Streams
module DnsProviders
# Hetzner Cloud DNS (Console) — zone mltv-stream.net.
#
# ENV:
# HCLOUD_TOKEN (stesso del Cloud)
# STREAM_DNS_ZONE (default mltv-stream.net)
# STREAM_DNS_TTL (default 60)
class Hetzner < Base
# Trailing slash obbligatorio: path assoluti altrimenti droppano /v1.
API = "https://api.hetzner.cloud/v1/"
def initialize(token: ENV.fetch("HCLOUD_TOKEN"), zone: nil, conn: nil)
@token = token
@zone = zone || ENV.fetch("STREAM_DNS_ZONE", "mltv-stream.net")
@ttl = ENV.fetch("STREAM_DNS_TTL", "60").to_i
@conn = conn
end
def upsert_a(name, ip)
rr_name = relative_name(name)
delete_a(name)
post("zones/#{CGI.escape(@zone)}/rrsets", {
name: rr_name,
type: "A",
ttl: @ttl,
records: [{ value: ip.to_s, comment: "matchlivetv stream-node" }],
labels: { "matchlivetv" => "true", "role" => "stream-node" }
})
true
end
def delete_a(name)
rr_name = relative_name(name)
encoded = CGI.escape(rr_name)
response = conn.delete("zones/#{CGI.escape(@zone)}/rrsets/#{encoded}/A") do |req|
req.headers.update(auth_headers)
end
return true if response.status == 404 || response.status == 204 || response.success?
raise Error, "Hetzner DNS API #{response.status}: #{response.body.inspect}"
end
def resolve(name)
rr_name = relative_name(name)
data = get("zones/#{CGI.escape(@zone)}/rrsets", name: rr_name, type: "A")
rrset = Array(data["rrsets"]).first
Array(rrset&.dig("records")).first&.dig("value")
rescue Error
nil
end
private
def relative_name(name)
host = name.to_s.strip.downcase.delete_suffix(".")
suffix = ".#{@zone}"
return "@" if host == @zone
return host.delete_suffix(suffix) if host.end_with?(suffix)
host
end
def conn
@conn ||= Faraday.new(url: API) do |f|
f.request :json
f.response :json, content_type: /\bjson$/
f.adapter Faraday.default_adapter
end
end
def auth_headers
{ "Authorization" => "Bearer #{@token}" }
end
def get(path, params = {})
response = conn.get(path) do |req|
req.headers.update(auth_headers)
req.params.update(params)
end
unwrap!(response)
end
def post(path, body)
response = conn.post(path) do |req|
req.headers.update(auth_headers)
req.body = body
end
unwrap!(response)
end
def unwrap!(response)
unless response.success?
raise Error, "Hetzner DNS API #{response.status}: #{response.body.inspect}"
end
response.body.is_a?(Hash) ? response.body : {}
end
end
end
end
@@ -0,0 +1,47 @@
# frozen_string_literal: true
module Streams
module DnsProviders
# DNS lab in Redis (e dump hosts). Nessuna chiamata al registrar.
class Lab < Base
REDIS_KEY = "stream_dns:a_records"
def initialize(redis: nil)
@redis = redis
end
def upsert_a(name, ip)
host = normalize(name)
redis.hset(REDIS_KEY, host, ip.to_s)
true
end
def delete_a(name)
redis.hdel(REDIS_KEY, normalize(name))
true
end
def resolve(name)
redis.hget(REDIS_KEY, normalize(name))
end
def all_records
redis.hgetall(REDIS_KEY)
end
def hosts_file_snippet
all_records.sort.map { |host, ip| "#{ip}\t#{host}" }.join("\n")
end
private
def normalize(name)
name.to_s.strip.downcase.delete_suffix(".")
end
def redis
@redis ||= Redis.new(url: ENV.fetch("REDIS_URL", "redis://localhost:6379/0"))
end
end
end
end
@@ -0,0 +1,152 @@
# frozen_string_literal: true
module Streams
# Provisiona / decommissiona nodi stream (lab o cloud) e aggiorna DNS + registry.
class NodeProvisioner
class Error < StandardError; end
class BusyError < Error; end
LAB_DNS_SUFFIX = -> { ENV.fetch("STREAM_LAB_DNS_SUFFIX", "lab.mltv-stream.net") }
CLOUD_DNS_SUFFIX = -> { ENV.fetch("STREAM_CLOUD_DNS_SUFFIX", ENV.fetch("STREAM_DNS_ZONE", "mltv-stream.net")) }
def initialize(cloud: nil, dns: nil)
@cloud = cloud
@dns = dns
end
def provision_lab!(prefix: "ingest-lab")
provision!(
prefix: prefix,
role: "lab",
dns_suffix: LAB_DNS_SUFFIX.call,
cloud: cloud_provider(ENV.fetch("STREAM_CLOUD_PROVIDER", "local_lab")),
dns: dns_provider(ENV.fetch("STREAM_DNS_PROVIDER", "lab")),
max: ENV.fetch("STREAM_LAB_MAX_PUBLISHERS", "2").to_i,
use_node_hostname: ENV["STREAM_LAB_USE_NODE_HOSTNAME"] == "1"
)
end
def provision_cloud!(prefix: "ingest")
provision!(
prefix: prefix,
role: "cloud",
dns_suffix: CLOUD_DNS_SUFFIX.call,
cloud: cloud_provider("hetzner"),
dns: dns_provider("hetzner"),
max: ENV.fetch("STREAM_CLOUD_MAX_PUBLISHERS", "4").to_i,
use_node_hostname: true
)
end
def decommission!(node)
raise Error, "Non si può decommissionare il nodo home" if node.slug == Streams::NodeRegistry::HOME_SLUG
if node.occupying_sessions.exists?
raise BusyError, "Nodo #{node.slug} ha ancora sessioni attive"
end
node.update!(status: "draining")
cloud = cloud_for_node(node)
dns = dns_for_node(node)
cloud.destroy_node(node.provider_instance_id) if node.provider_instance_id.present?
dns.delete_a(node.hostname) if node.hostname.present?
node.destroy!
true
end
def drain!(node)
raise Error, "Non si può mettere in drain il nodo home" if node.slug == Streams::NodeRegistry::HOME_SLUG
node.update!(status: "draining")
node
end
private
def provision!(prefix:, role:, dns_suffix:, cloud:, dns:, max:, use_node_hostname:)
Streams::NodeRegistry.ensure_home_from_env!
home = StreamNode.find_by!(slug: Streams::NodeRegistry::HOME_SLUG)
slug = next_slug(prefix)
hostname = "#{slug}.#{dns_suffix}"
instance = cloud.create_node(
name: "mltv-stream-#{slug}",
labels: { role: "stream-node", env: role == "cloud" ? "prod" : "lab" }
)
ip = instance.public_ip.presence || "127.0.0.1"
private_ip = instance.private_ip.presence || ip
dns.upsert_a(hostname, ip)
simulated = instance.raw.is_a?(Hash) && (instance.raw[:simulated] || instance.raw["simulated"])
api_base = simulated ? home.api_base_url : "http://#{private_ip}:9997"
internal_rtmp = simulated ? home.internal_rtmp_url : "rtmp://#{private_ip}:1935"
internal_hls = simulated ? home.internal_hls_url : "http://#{private_ip}:8888"
StreamNode.create!(
slug: slug,
hostname: hostname,
role: role,
status: "ready",
provider: provider_name_for(cloud, role: role),
provider_instance_id: instance.id,
rtmp_base_url: use_node_hostname ? "rtmp://#{hostname}:1935" : home.rtmp_base_url,
hls_base_url: use_node_hostname ? "https://#{hostname}/hls" : home.hls_base_url,
api_base_url: api_base,
internal_rtmp_url: internal_rtmp,
internal_hls_url: internal_hls,
max_publishers: max,
max_relays: max,
last_health_at: Time.current,
metadata: {
"public_ip" => ip,
"private_ip" => private_ip,
"simulated" => simulated,
"cloud_raw" => instance.raw
}
)
end
def next_slug(prefix)
used = StreamNode.where("slug LIKE ?", "#{prefix}-%").pluck(:slug)
n = 1
loop do
candidate = format("%s-%02d", prefix, n)
return candidate unless used.include?(candidate)
n += 1
end
end
def cloud_provider(name)
@cloud || Streams::CloudProviders.build(name)
end
def dns_provider(name)
@dns || Streams::DnsProviders.build(name)
end
def provider_name_for(cloud, role: nil)
return "hetzner" if role.to_s == "cloud"
case cloud
when Streams::CloudProviders::ProxmoxLab then "proxmox_lab"
when Streams::CloudProviders::Hetzner then "hetzner"
else "local"
end
end
def cloud_for_node(node)
case node.provider
when "hetzner" then Streams::CloudProviders::Hetzner.new
when "proxmox_lab" then Streams::CloudProviders::ProxmoxLab.new
else Streams::CloudProviders::LocalLab.new
end
end
def dns_for_node(node)
case node.provider
when "hetzner" then Streams::DnsProviders::Hetzner.new
else Streams::DnsProviders::Lab.new
end
end
end
end
@@ -0,0 +1,59 @@
# frozen_string_literal: true
module Streams
# Assegna un StreamNode a una nuova sessione (least-loaded tra i ready).
# Garantisce il nodo "home" derivato dagli ENV MediaMTX attuali.
class NodeRegistry
class NoCapacityError < StandardError; end
HOME_SLUG = "home"
class << self
def ensure_home_from_env!
StreamNode.find_or_initialize_by(slug: HOME_SLUG).tap do |node|
attrs = {
hostname: home_hostname,
role: "home",
provider: "local",
rtmp_base_url: MatchLiveTv.mediamtx_rtmp_url,
hls_base_url: MatchLiveTv.hls_public_url,
api_base_url: MatchLiveTv.mediamtx_api_url,
internal_rtmp_url: ENV.fetch("MEDIAMTX_INTERNAL_RTMP_URL", "rtmp://mediamtx:1935"),
internal_hls_url: ENV.fetch("MEDIAMTX_HLS_URL", "http://mediamtx:8888"),
max_publishers: ENV.fetch("STREAM_NODE_HOME_MAX_PUBLISHERS", "6").to_i,
max_relays: ENV.fetch("STREAM_NODE_HOME_MAX_RELAYS", "6").to_i
}
# Non sovrascrivere drain/offline/error (ops) a ogni allocate!
attrs[:status] = "ready" if node.new_record? || !%w[draining offline error].include?(node.status)
node.assign_attributes(attrs)
node.save!
end
end
def allocate!
ensure_home_from_env!
node = StreamNode.ready
.to_a
.select(&:allocatable?)
.min_by { |n| [n.active_publishers, n.role == "home" ? 1 : 0, n.slug] }
raise NoCapacityError, "Nessun nodo streaming con slot liberi" if node.nil?
node
end
private
def home_hostname
ENV["STREAM_NODE_HOME_HOSTNAME"].presence ||
begin
uri = URI.parse(MatchLiveTv.mediamtx_rtmp_url.sub(/\Artmps?:\/\//, "http://"))
uri.host.presence
rescue URI::InvalidURIError
nil
end || "home"
end
end
end
end
+83 -23
View File
@@ -1,29 +1,35 @@
module Streams
# Relay verso YouTube: legge RTMP/HLS da MediaMTX e inoltra su RTMPS (-c copy). Nessun overlay.
# ffmpeg gira solo nel container sidekiq (YOUTUBE_RELAY_WORKER=1).
# ffmpeg gira solo sui worker Sidekiq con YOUTUBE_RELAY_WORKER=1 (coda youtube_relay).
class YoutubeRelay
class Error < StandardError; end
REDIS_KEY = "youtube_relay:pid:%s"
OWNER_KEY = "youtube_relay:owner:%s"
OWNED_SET = "youtube_relay:owned:%s"
QUEUE = :youtube_relay
class << self
def worker?
ENV["YOUTUBE_RELAY_WORKER"] == "1"
end
def max_concurrent
ENV.fetch("RELAY_MAX_CONCURRENT", "4").to_i
end
def start(session)
return unless worker?
start_on_worker!(session)
end
# Non cancella owner/pid qui: solo il worker owner deve killare ffmpeg.
def stop(session)
clear_pid(session.id)
redis.del(format(OWNER_KEY, session.id))
if worker?
if worker? && owner_is_local?(session.id)
stop_on_worker!(session)
else
YoutubeRelayStopJob.perform_later(session.id)
YoutubeRelayStopJob.set(queue: QUEUE).perform_later(session.id)
end
true
end
@@ -31,11 +37,13 @@ module Streams
def running?(session_id)
owner = redis.get(format(OWNER_KEY, session_id))
pid = pid_for(session_id)
return false if pid.blank? && owner.blank?
return false if pid.blank? && owner.present? && owner != worker_id && redis.ttl(format(OWNER_KEY, session_id)) <= 30
return false if pid.blank?
return process_alive?(pid) if owner.blank? || owner == worker_id
# Relay avviato in un altro container: consideralo attivo se il lock è recente.
# Relay su altro host: attivo se lock owner ancora fresco.
redis.ttl(format(OWNER_KEY, session_id)) > 30
end
@@ -47,12 +55,12 @@ module Streams
if worker?
ensure_on_worker!(session)
else
YoutubeRelayEnsureJob.perform_later(session.id)
YoutubeRelayEnsureJob.set(queue: QUEUE).perform_later(session.id)
end
end
def ensure_on_worker!(session)
return unless worker?
return :not_worker unless worker?
return unless session.platform == "youtube"
return if session.terminal?
return if session.stream_key.blank?
@@ -60,28 +68,64 @@ module Streams
return unless intake_available?(session)
pid = pid_for(session.id)
clear_pid(session.id) if pid.present? && !process_alive?(pid.to_i)
if pid.present? && !process_alive?(pid.to_i)
clear_local_ownership(session.id)
end
return if running?(session.id)
if running?(session.id)
touch_owner!(session.id) if owner_is_local?(session.id)
return :already_running
end
if at_capacity?
YoutubeRelayEnsureJob.set(wait: 5.seconds, queue: QUEUE).perform_later(session.id)
Rails.logger.info("[YoutubeRelay] at capacity worker=#{worker_id} session=#{session.id} requeue")
return :at_capacity
end
last_restart = redis.get(restart_debounce_key(session.id)).to_i
return if last_restart.positive? && (Time.now.to_i - last_restart) < 5
return :debounced if last_restart.positive? && (Time.now.to_i - last_restart) < 5
start_on_worker!(session)
redis.set(restart_debounce_key(session.id), Time.now.to_i, ex: 300)
:started
rescue Error => e
Rails.logger.warn("[YoutubeRelay] ensure_on_worker session=#{session.id}: #{e.message}")
:error
end
# @return [Symbol] :stopped, :wrong_host, :noop
def stop_on_worker!(session)
return :not_worker unless worker?
owner = redis.get(format(OWNER_KEY, session.id))
if owner.present? && owner != worker_id
return :wrong_host
end
pid = pid_for(session.id)
return false if pid.blank?
if pid.blank?
clear_local_ownership(session.id)
return :noop
end
terminate_pid(pid)
clear_pid(session.id)
redis.del(format(OWNER_KEY, session.id))
Rails.logger.info("[YoutubeRelay] stopped pid=#{pid} session=#{session.id}")
true
clear_local_ownership(session.id)
Rails.logger.info("[YoutubeRelay] stopped pid=#{pid} session=#{session.id} worker=#{worker_id}")
:stopped
end
def local_owned_count
redis.scard(format(OWNED_SET, worker_id)).to_i
end
def at_capacity?
local_owned_count >= max_concurrent
end
def owner_is_local?(session_id)
owner = redis.get(format(OWNER_KEY, session_id))
owner.blank? || owner == worker_id
end
private
@@ -92,9 +136,9 @@ module Streams
return if session.terminal?
return unless intake_available?(session)
return pid_for(session.id).to_i if running?(session.id)
return pid_for(session.id).to_i if running?(session.id) && owner_is_local?(session.id)
stop_on_worker!(session) if pid_for(session.id).present?
stop_on_worker!(session) if pid_for(session.id).present? && owner_is_local?(session.id)
log_path = log_file(session)
FileUtils.mkdir_p(File.dirname(log_path))
@@ -108,8 +152,8 @@ module Streams
)
Process.detach(pid)
store_pid(session.id, pid)
redis.set(format(OWNER_KEY, session.id), worker_id, ex: 48.hours.to_i)
Rails.logger.info("[YoutubeRelay] started pid=#{pid} session=#{session.id} intake=#{intake_source.join(":")}")
claim_ownership!(session.id)
Rails.logger.info("[YoutubeRelay] started pid=#{pid} session=#{session.id} intake=#{intake_source.join(":")} worker=#{worker_id}")
schedule_youtube_activate(session)
pid
rescue Errno::ENOENT => e
@@ -146,20 +190,20 @@ module Streams
end
def mediamtx_intake_source(session)
base = ENV.fetch("MEDIAMTX_INTERNAL_RTMP_URL", "rtmp://mediamtx:1935")
base = session.mediamtx_internal_rtmp_url
if Mediamtx::PublisherOnline.active?(session)
return [:rtmp, "#{base.chomp('/')}/#{session.mediamtx_path_name}"]
end
path = session.mediamtx_path_name
hls = ENV.fetch("MEDIAMTX_HLS_URL", "http://mediamtx:8888").chomp("/")
hls = session.mediamtx_internal_hls_url.chomp("/")
[:hls, "#{hls}/#{path}/index.m3u8"]
end
def intake_available?(session)
return true if Mediamtx::PublisherOnline.active?(session)
info = Mediamtx::Client.new.list_paths.find { |i| i["name"] == session.mediamtx_path_name }
info = Mediamtx::Client.for_session(session).list_paths.find { |i| i["name"] == session.mediamtx_path_name }
info && (info["ready"] || info["online"] || info["available"])
rescue StandardError
false
@@ -185,6 +229,22 @@ module Streams
format("youtube_relay:debounce:%s", session_id)
end
def claim_ownership!(session_id)
redis.set(format(OWNER_KEY, session_id), worker_id, ex: 48.hours.to_i)
redis.sadd(format(OWNED_SET, worker_id), session_id)
end
def touch_owner!(session_id)
redis.expire(format(OWNER_KEY, session_id), 48.hours.to_i)
redis.expire(format(REDIS_KEY, session_id), 48.hours.to_i)
end
def clear_local_ownership(session_id)
redis.del(format(REDIS_KEY, session_id))
redis.del(format(OWNER_KEY, session_id))
redis.srem(format(OWNED_SET, worker_id), session_id)
end
def store_pid(session_id, pid)
redis.set(format(REDIS_KEY, session_id), pid, ex: 48.hours.to_i)
end
@@ -68,13 +68,13 @@ module Webhooks
def enable_recording(session)
return unless session.match.team.entitlements.recording_enabled_for_mediamtx?
Mediamtx::Client.new.set_path_recording(session, enabled: true)
Mediamtx::Client.for_session(session).set_path_recording(session, enabled: true)
rescue Mediamtx::Client::Error => e
Rails.logger.warn("[MediamtxHandler] enable recording: #{e.message}")
end
def disable_recording(session)
Mediamtx::Client.new.set_path_recording(session, enabled: false)
Mediamtx::Client.for_session(session).set_path_recording(session, enabled: false)
rescue Mediamtx::Client::Error => e
Rails.logger.warn("[MediamtxHandler] disable recording: #{e.message}")
end
@@ -63,7 +63,7 @@ module Youtube
return
end
Mediamtx::Client.new.set_always_available(session, enabled: false)
Mediamtx::Client.for_session(session).set_always_available(session, enabled: false)
session.go_live! if session.may_go_live?
session.reconnect! if session.reconnecting? && session.may_reconnect?
@@ -0,0 +1,83 @@
<% content_for :body_class, "admin-body" %>
<h2><%= t("admin.stream_nodes.title") %></h2>
<p class="admin-muted">
<%= t("admin.stream_nodes.providers", cloud: @cloud_provider, dns: @dns_provider) %>
</p>
<p class="admin-muted">
<%= t(
"admin.stream_nodes.autoscale",
enabled: (@autoscale_metrics[:enabled] ? "ON" : "OFF"),
free: @autoscale_metrics[:free_slots],
soft: @autoscale_metrics[:soft_free_slots],
spare: @autoscale_metrics[:spare_ready],
warm: @autoscale_metrics[:warm_spare_min],
overflow: @autoscale_metrics[:overflow_nodes],
max: @autoscale_metrics[:max_overflow_nodes],
kind: @autoscale_metrics[:kind]
) %>
· <%= t(
"admin.stream_nodes.autoscale_budget",
eur: @autoscale_metrics[:estimated_monthly_eur],
budget: @autoscale_metrics[:monthly_budget_eur],
ok: (@autoscale_metrics[:within_budget] ? "OK" : "OVER")
) %>
<% if @autoscale_metrics[:kill_switch] %>
· <strong><%= t("admin.stream_nodes.kill_switch_active") %></strong>
<% end %>
</p>
<p>
<%= button_to t("admin.stream_nodes.provision_lab"), admin_stream_nodes_path, method: :post, params: { kind: "lab" }, class: "admin-btn" %>
<% if @hetzner_configured %>
<%= button_to t("admin.stream_nodes.provision_cloud"), admin_stream_nodes_path, method: :post, params: { kind: "cloud" }, class: "admin-btn",
form: { data: { confirm: t("admin.stream_nodes.provision_cloud_confirm") } } %>
<% else %>
<span class="admin-muted"><%= t("admin.stream_nodes.hetzner_token_missing") %></span>
<% end %>
<% if @autoscale_metrics[:kill_switch] %>
<%= button_to t("admin.stream_nodes.clear_kill_switch"), clear_kill_switch_admin_stream_nodes_path, method: :delete, class: "admin-btn admin-btn-secondary" %>
<% else %>
<%= button_to t("admin.stream_nodes.engage_kill_switch"), kill_switch_admin_stream_nodes_path, method: :post, class: "admin-btn admin-btn-danger",
form: { data: { confirm: t("admin.stream_nodes.kill_switch_confirm") } } %>
<% end %>
</p>
<table class="admin-table">
<thead>
<tr>
<th><%= t("admin.stream_nodes.col.slug") %></th>
<th><%= t("admin.stream_nodes.col.role") %></th>
<th><%= t("admin.stream_nodes.col.status") %></th>
<th><%= t("admin.stream_nodes.col.slots") %></th>
<th><%= t("admin.stream_nodes.col.hostname") %></th>
<th><%= t("admin.stream_nodes.col.provider") %></th>
<th></th>
</tr>
</thead>
<tbody>
<% @nodes.each do |node| %>
<tr>
<td><code><%= node.slug %></code></td>
<td><%= node.role %></td>
<td><%= node.status %></td>
<td><%= node.active_publishers %> / <%= node.max_publishers %> (free <%= node.free_slots %>)</td>
<td><code><%= node.hostname %></code></td>
<td><%= node.provider %><% if node.provider_instance_id.present? %> · <%= node.provider_instance_id %><% end %></td>
<td class="admin-actions">
<% if node.slug != "home" %>
<% if node.status == "ready" %>
<%= button_to t("admin.stream_nodes.drain"), drain_admin_stream_node_path(node), method: :post, class: "admin-btn admin-btn-secondary" %>
<% end %>
<%= button_to t("admin.stream_nodes.destroy"), admin_stream_node_path(node), method: :delete, class: "admin-btn admin-btn-danger", form: { data: { confirm: t("admin.stream_nodes.destroy_confirm", slug: node.slug) } } %>
<% end %>
</td>
</tr>
<% end %>
</tbody>
</table>
<% if @lab_hosts.present? %>
<h3><%= t("admin.stream_nodes.hosts_title") %></h3>
<pre class="admin-pre"><%= @lab_hosts %></pre>
<% end %>
+1
View File
@@ -29,6 +29,7 @@
<%= link_to t("admin.layout.nav.billing"), admin_billing_path, class: ("active" if controller_name.in?(%w[billing billing_invoices])) %>
<%= link_to t("admin.layout.nav.youtube"), admin_youtube_platform_path, class: ("active" if controller_name == "youtube") %>
<%= link_to t("admin.layout.nav.sessions"), admin_sessions_path, class: ("active" if controller_name == "sessions") %>
<%= link_to t("admin.layout.nav.stream_nodes"), admin_stream_nodes_path, class: ("active" if controller_name == "stream_nodes") %>
<%= link_to t("admin.layout.nav.password"), edit_admin_password_path %>
<%= button_to t("admin.layout.nav.logout"), admin_logout_path, method: :delete %>
<% end %>
+1
View File
@@ -6,6 +6,7 @@ Sidekiq.configure_server do |config|
config.on(:startup) do
StreamPublisherSyncJob.ensure_chain
Ops::HealthMonitorJob.ensure_chain
Streams::AutoscalerJob.ensure_chain
end
end
+27
View File
@@ -9,6 +9,7 @@ de:
billing: Abrechnung
youtube: YouTube
sessions: Sitzungen
stream_nodes: Stream-Knoten
password: Passwort
logout: Abmelden
flash:
@@ -26,6 +27,11 @@ de:
ops_acknowledged: Vorfall übernommen
ops_resolved: Vorfall gelöst
ops_muted: Benachrichtigungen für 24 Stunden stummgeschaltet
stream_node_created: "Lab-Knoten %{slug} bereitgestellt."
stream_node_destroyed: "Knoten %{slug} entfernt."
stream_node_draining: "Knoten %{slug} im Drain-Modus."
autoscale_kill_on: "Autoscaler-Kill-Switch aktiv. Kein automatisches Scale-out."
autoscale_kill_off: "Autoscaler-Kill-Switch aus (STREAM_AUTOSCALE_ENABLED=1 weiterhin nötig)."
comped_granted: "Kostenloses Abonnement %{plan} für %{club} aktiviert."
comped_revoked: "Kostenloses Abonnement für %{club} widerrufen."
session_already_terminated: "Sitzung bereits beendet (%{status})."
@@ -111,6 +117,27 @@ de:
title: Teams
matches_count: "%{count} Spiele"
view_all: "Vereine ansehen (%{count} Teams)"
stream_nodes:
title: Stream-Knoten
providers: "Cloud-Provider: %{cloud} · DNS-Provider: %{dns}"
autoscale: "Autoscaler %{enabled} · free_slots=%{free} (soft≤%{soft}) · spare=%{spare}/%{warm} · overflow=%{overflow}/%{max} · kind=%{kind}"
autoscale_budget: "Budget €%{eur}/€%{budget} (%{ok})"
kill_switch_active: "KILL-SWITCH AKTIV"
engage_kill_switch: "Kill-switch ON"
clear_kill_switch: "Kill-switch OFF"
kill_switch_confirm: "Autoscaler sofort blockieren?"
provision_lab: Lab-Knoten bereitstellen
drain: Drain
destroy: Löschen
destroy_confirm: "Knoten %{slug} löschen?"
hosts_title: Lab-DNS (/etc/hosts)
col:
slug: Slug
role: Rolle
status: Status
slots: Slots
hostname: Hostname
provider: Provider
ops:
kpi:
critical: Kritisch offen
+30
View File
@@ -9,6 +9,7 @@ en:
billing: Billing
youtube: YouTube
sessions: Sessions
stream_nodes: Stream nodes
password: Password
logout: Log out
flash:
@@ -26,6 +27,11 @@ en:
ops_acknowledged: Incident acknowledged
ops_resolved: Incident resolved
ops_muted: Notifications muted for 24 hours
stream_node_created: "Lab node %{slug} provisioned."
stream_node_destroyed: "Node %{slug} removed."
stream_node_draining: "Node %{slug} is draining (no new sessions)."
autoscale_kill_on: "Autoscaler kill-switch engaged. No automatic scale-out."
autoscale_kill_off: "Autoscaler kill-switch cleared (still needs STREAM_AUTOSCALE_ENABLED=1)."
comped_granted: "%{plan} complimentary subscription activated for %{club}."
comped_revoked: "Complimentary subscription revoked for %{club}."
session_already_terminated: "Session already ended (%{status})."
@@ -111,6 +117,30 @@ en:
title: Teams
matches_count: "%{count} matches"
view_all: "View clubs (%{count} teams)"
stream_nodes:
title: Streaming nodes
providers: "Cloud provider: %{cloud} · DNS provider: %{dns}"
autoscale: "Autoscaler %{enabled} · free_slots=%{free} (soft≤%{soft}) · spare=%{spare}/%{warm} · overflow=%{overflow}/%{max} · kind=%{kind}"
autoscale_budget: "budget €%{eur}/€%{budget} (%{ok})"
kill_switch_active: "KILL-SWITCH ACTIVE"
engage_kill_switch: "Kill-switch ON (block autoscaler)"
clear_kill_switch: "Kill-switch OFF"
kill_switch_confirm: "Immediately block the autoscaler? Existing nodes stay up."
provision_lab: Provision lab node
provision_cloud: Provision Hetzner node
provision_cloud_confirm: "Create a Hetzner Cloud server + DNS record on mltv-stream.net? Billing applies until destroyed."
hetzner_token_missing: "Set HCLOUD_TOKEN to enable Cloud provisioning."
drain: Drain
destroy: Delete
destroy_confirm: "Delete node %{slug}?"
hosts_title: Lab DNS records (/etc/hosts snippet)
col:
slug: Slug
role: Role
status: Status
slots: Slots
hostname: Hostname
provider: Provider
ops:
kpi:
critical: Critical open
+27
View File
@@ -9,6 +9,7 @@ es:
billing: Facturación
youtube: YouTube
sessions: Sesiones
stream_nodes: Nodos stream
password: Contraseña
logout: Salir
flash:
@@ -26,6 +27,11 @@ es:
ops_acknowledged: Incidencia asumida
ops_resolved: Incidencia resuelta
ops_muted: Notificaciones silenciadas durante 24 horas
stream_node_created: "Nodo lab %{slug} provisionado."
stream_node_destroyed: "Nodo %{slug} eliminado."
stream_node_draining: "Nodo %{slug} en drain."
autoscale_kill_on: "Kill-switch del autoscaler activado. Sin scale-out automático."
autoscale_kill_off: "Kill-switch del autoscaler desactivado (hace falta STREAM_AUTOSCALE_ENABLED=1)."
comped_granted: "Suscripción de cortesía %{plan} activada para %{club}."
comped_revoked: "Suscripción de cortesía revocada para %{club}."
session_already_terminated: "La sesión ya ha finalizado (%{status})."
@@ -111,6 +117,27 @@ es:
title: Equipos
matches_count: "%{count} partidos"
view_all: "Ver clubes (%{count} equipos)"
stream_nodes:
title: Nodos streaming
providers: "Cloud provider: %{cloud} · DNS provider: %{dns}"
autoscale: "Autoscaler %{enabled} · free_slots=%{free} (soft≤%{soft}) · spare=%{spare}/%{warm} · overflow=%{overflow}/%{max} · kind=%{kind}"
autoscale_budget: "presupuesto €%{eur}/€%{budget} (%{ok})"
kill_switch_active: "KILL-SWITCH ACTIVO"
engage_kill_switch: "Kill-switch ON"
clear_kill_switch: "Kill-switch OFF"
kill_switch_confirm: "¿Bloquear el autoscaler inmediatamente?"
provision_lab: Provisionar nodo lab
drain: Drain
destroy: Eliminar
destroy_confirm: "¿Eliminar el nodo %{slug}?"
hosts_title: DNS lab (/etc/hosts)
col:
slug: Slug
role: Rol
status: Estado
slots: Slots
hostname: Hostname
provider: Provider
ops:
kpi:
critical: Críticas abiertas
+27
View File
@@ -9,6 +9,7 @@ fr:
billing: Facturation
youtube: YouTube
sessions: Sessions
stream_nodes: Nœuds stream
password: Mot de passe
logout: Déconnexion
flash:
@@ -26,6 +27,11 @@ fr:
ops_acknowledged: Incident pris en charge
ops_resolved: Incident résolu
ops_muted: Notifications suspendues pendant 24 heures
stream_node_created: "Nœud lab %{slug} provisionné."
stream_node_destroyed: "Nœud %{slug} supprimé."
stream_node_draining: "Nœud %{slug} en drain."
autoscale_kill_on: "Kill-switch autoscaler activé. Pas de scale-out automatique."
autoscale_kill_off: "Kill-switch autoscaler désactivé (nécessite aussi STREAM_AUTOSCALE_ENABLED=1)."
comped_granted: "Abonnement offert %{plan} activé pour %{club}."
comped_revoked: "Abonnement offert révoqué pour %{club}."
session_already_terminated: "Session déjà terminée (%{status})."
@@ -111,6 +117,27 @@ fr:
title: Équipes
matches_count: "%{count} matchs"
view_all: "Voir les clubs (%{count} équipes)"
stream_nodes:
title: Nœuds streaming
providers: "Cloud provider: %{cloud} · DNS provider: %{dns}"
autoscale: "Autoscaler %{enabled} · free_slots=%{free} (soft≤%{soft}) · spare=%{spare}/%{warm} · overflow=%{overflow}/%{max} · kind=%{kind}"
autoscale_budget: "budget €%{eur}/€%{budget} (%{ok})"
kill_switch_active: "KILL-SWITCH ACTIF"
engage_kill_switch: "Kill-switch ON"
clear_kill_switch: "Kill-switch OFF"
kill_switch_confirm: "Bloquer immédiatement l'autoscaler ?"
provision_lab: Provisionner un nœud lab
drain: Drain
destroy: Supprimer
destroy_confirm: "Supprimer le nœud %{slug} ?"
hosts_title: DNS lab (/etc/hosts)
col:
slug: Slug
role: Rôle
status: Statut
slots: Slots
hostname: Hostname
provider: Provider
ops:
kpi:
critical: Critiques ouverts
+30
View File
@@ -9,6 +9,7 @@ it:
billing: Fatturazione
youtube: YouTube
sessions: Sessioni
stream_nodes: Nodi stream
password: Password
logout: Esci
flash:
@@ -26,6 +27,11 @@ it:
ops_acknowledged: Incidente preso in carico
ops_resolved: Incidente risolto
ops_muted: Notifiche sospese per 24 ore
stream_node_created: "Nodo lab %{slug} provisionato."
stream_node_destroyed: "Nodo %{slug} rimosso."
stream_node_draining: "Nodo %{slug} in drain (niente nuove sessioni)."
autoscale_kill_on: "Kill-switch autoscaler attivato. Nessun scale-out automatico."
autoscale_kill_off: "Kill-switch autoscaler disattivato (serve comunque STREAM_AUTOSCALE_ENABLED=1)."
comped_granted: "Abbonamento omaggio %{plan} attivato per %{club}."
comped_revoked: "Abbonamento omaggio revocato per %{club}."
session_already_terminated: "Sessione già terminata (%{status})."
@@ -111,6 +117,30 @@ it:
title: Squadre
matches_count: "%{count} partite"
view_all: "Vedi società (%{count} squadre)"
stream_nodes:
title: Nodi streaming
providers: "Cloud provider: %{cloud} · DNS provider: %{dns}"
autoscale: "Autoscaler %{enabled} · free_slots=%{free} (soft≤%{soft}) · spare=%{spare}/%{warm} · overflow=%{overflow}/%{max} · kind=%{kind}"
autoscale_budget: "budget €%{eur}/€%{budget} (%{ok})"
kill_switch_active: "KILL-SWITCH ATTIVO"
engage_kill_switch: "Kill-switch ON (blocca autoscaler)"
clear_kill_switch: "Kill-switch OFF"
kill_switch_confirm: "Bloccare immediatamente l'autoscaler? I nodi esistenti restano accesi."
provision_lab: Provisiona nodo lab
provision_cloud: Provisiona nodo Hetzner
provision_cloud_confirm: "Creare un server Hetzner Cloud + record DNS su mltv-stream.net? Verrà addebitato fino allo spegnimento."
hetzner_token_missing: "Imposta HCLOUD_TOKEN per abilitare il provisioning Cloud."
drain: Drain
destroy: Elimina
destroy_confirm: "Eliminare il nodo %{slug}?"
hosts_title: Record DNS lab (snippet /etc/hosts)
col:
slug: Slug
role: Ruolo
status: Stato
slots: Slot
hostname: Hostname
provider: Provider
ops:
kpi:
critical: Critici aperti
+9
View File
@@ -110,6 +110,15 @@ Rails.application.routes.draw do
post :regia_link
end
end
resources :stream_nodes, only: %i[index create destroy] do
member do
post :drain
end
collection do
post :kill_switch
delete :clear_kill_switch
end
end
get "youtube/platform", to: "youtube#platform", as: :youtube_platform
end
+2 -1
View File
@@ -1,4 +1,5 @@
:concurrency: 5
:queues:
- default
- critical
- youtube_relay
- default
@@ -0,0 +1,30 @@
# frozen_string_literal: true
class CreateStreamNodes < ActiveRecord::Migration[7.2]
def change
create_table :stream_nodes, id: :uuid, default: -> { "gen_random_uuid()" } do |t|
t.string :slug, null: false
t.string :hostname, null: false
t.string :role, null: false, default: "home"
t.string :status, null: false, default: "ready"
t.string :rtmp_base_url, null: false
t.string :hls_base_url, null: false
t.string :api_base_url, null: false
t.string :internal_rtmp_url
t.string :internal_hls_url
t.integer :max_publishers, null: false, default: 6
t.integer :max_relays, null: false, default: 6
t.string :provider, null: false, default: "local"
t.string :provider_instance_id
t.datetime :last_health_at
t.jsonb :metadata, null: false, default: {}
t.timestamps
end
add_index :stream_nodes, :slug, unique: true
add_index :stream_nodes, :status
add_index :stream_nodes, :role
add_reference :stream_sessions, :stream_node, type: :uuid, foreign_key: true, null: true, index: true
end
end
+27 -1
View File
@@ -10,7 +10,7 @@
#
# It's strongly recommended that you check this file into your version control system.
ActiveRecord::Schema[7.2].define(version: 2026_06_12_120000) do
ActiveRecord::Schema[7.2].define(version: 2026_08_09_120000) do
# These are extensions that must be enabled in order to support this database
enable_extension "pgcrypto"
enable_extension "plpgsql"
@@ -255,6 +255,29 @@ ActiveRecord::Schema[7.2].define(version: 2026_06_12_120000) do
t.index ["stream_session_id"], name: "index_stream_events_on_stream_session_id"
end
create_table "stream_nodes", id: :uuid, default: -> { "gen_random_uuid()" }, force: :cascade do |t|
t.string "slug", null: false
t.string "hostname", null: false
t.string "role", default: "home", null: false
t.string "status", default: "ready", null: false
t.string "rtmp_base_url", null: false
t.string "hls_base_url", null: false
t.string "api_base_url", null: false
t.string "internal_rtmp_url"
t.string "internal_hls_url"
t.integer "max_publishers", default: 6, null: false
t.integer "max_relays", default: 6, null: false
t.string "provider", default: "local", null: false
t.string "provider_instance_id"
t.datetime "last_health_at"
t.jsonb "metadata", default: {}, null: false
t.datetime "created_at", null: false
t.datetime "updated_at", null: false
t.index ["role"], name: "index_stream_nodes_on_role"
t.index ["slug"], name: "index_stream_nodes_on_slug", unique: true
t.index ["status"], name: "index_stream_nodes_on_status"
end
create_table "stream_sessions", id: :uuid, default: -> { "gen_random_uuid()" }, force: :cascade do |t|
t.uuid "match_id", null: false
t.uuid "user_id", null: false
@@ -280,10 +303,12 @@ ActiveRecord::Schema[7.2].define(version: 2026_06_12_120000) do
t.datetime "updated_at", null: false
t.string "regia_token_digest"
t.datetime "regia_token_expires_at"
t.uuid "stream_node_id"
t.index ["match_id"], name: "index_stream_sessions_on_match_id"
t.index ["publish_token"], name: "index_stream_sessions_on_publish_token", unique: true
t.index ["regia_token_digest"], name: "index_stream_sessions_on_regia_token_digest", unique: true
t.index ["status"], name: "index_stream_sessions_on_status"
t.index ["stream_node_id"], name: "index_stream_sessions_on_stream_node_id"
t.index ["user_id"], name: "index_stream_sessions_on_user_id"
end
@@ -411,6 +436,7 @@ ActiveRecord::Schema[7.2].define(version: 2026_06_12_120000) do
add_foreign_key "score_states", "stream_sessions"
add_foreign_key "stream_events", "stream_sessions"
add_foreign_key "stream_sessions", "matches"
add_foreign_key "stream_sessions", "stream_nodes"
add_foreign_key "stream_sessions", "users"
add_foreign_key "subscriptions", "admin_accounts", column: "admin_comped_by_id"
add_foreign_key "subscriptions", "clubs"
+3
View File
@@ -43,3 +43,6 @@ end
puts "Seed OK: coach@matchlivetv.test / Password123"
puts "Club: #{club.name}, Team: #{team.name}, Match: #{match.opponent_name}"
home = Streams::NodeRegistry.ensure_home_from_env!
puts "Stream node home: #{home.slug} rtmp=#{home.rtmp_base_url}"
+32
View File
@@ -0,0 +1,32 @@
# frozen_string_literal: true
namespace :streams do
namespace :nodes do
desc "Assicura il nodo home dagli ENV MediaMTX"
task ensure_home: :environment do
node = Streams::NodeRegistry.ensure_home_from_env!
puts "home ready slug=#{node.slug} rtmp=#{node.rtmp_base_url} max=#{node.max_publishers}"
end
desc "Provisiona un nodo lab (STREAM_CLOUD_PROVIDER=local_lab|proxmox_lab)"
task provision_lab: :environment do
node = Streams::NodeProvisioner.new.provision_lab!
puts "lab node ready slug=#{node.slug} host=#{node.hostname} id=#{node.provider_instance_id}"
if ENV.fetch("STREAM_DNS_PROVIDER", "lab") == "lab"
puts "DNS lab snippet:"
puts Streams::DnsProviders::Lab.new.hosts_file_snippet
end
end
desc "Provisiona un nodo Hetzner Cloud + DNS mltv-stream.net (richiede HCLOUD_TOKEN)"
task provision_cloud: :environment do
node = Streams::NodeProvisioner.new.provision_cloud!
puts "cloud node ready slug=#{node.slug} host=#{node.hostname} id=#{node.provider_instance_id} ip=#{node.metadata['public_ip']}"
end
desc "Dump record DNS lab (Redis)"
task dns_lab_dump: :environment do
puts Streams::DnsProviders::Lab.new.hosts_file_snippet
end
end
end
@@ -0,0 +1,18 @@
# frozen_string_literal: true
require "rails_helper"
RSpec.describe YoutubeRelayStopJob, type: :job do
it "requeues when stop runs on the wrong host" do
session = instance_double(StreamSession, id: SecureRandom.uuid)
allow(StreamSession).to receive(:find_by).and_return(session)
allow(Streams::YoutubeRelay).to receive(:worker?).and_return(true)
allow(Streams::YoutubeRelay).to receive(:stop_on_worker!).and_return(:wrong_host)
job_proxy = double("ConfiguredJob")
expect(described_class).to receive(:set).with(wait: 2.seconds).and_return(job_proxy)
expect(job_proxy).to receive(:perform_later).with(session.id, 1)
described_class.new.perform(session.id, 0)
end
end
@@ -0,0 +1,37 @@
# frozen_string_literal: true
require "rails_helper"
RSpec.describe StreamSession do
it "builds ingest/hls URLs from the assigned stream node" do
node = StreamNode.create!(
slug: "ingest-01",
hostname: "ingest-01.mltv-stream.net",
role: "cloud",
status: "ready",
provider: "hetzner",
rtmp_base_url: "rtmp://ingest-01.mltv-stream.net:1935",
hls_base_url: "https://ingest-01.mltv-stream.net/hls",
api_base_url: "http://10.0.0.2:9997",
internal_rtmp_url: "rtmp://10.0.0.2:1935",
max_publishers: 4,
max_relays: 4
)
user = User.create!(email: "s@example.com", name: "S", password: "Password123", role: "coach")
club = Club.create!(name: "Club", sport: "volleyball")
team = club.teams.create!(name: "Team", sport: "volleyball", slug: "team-url")
match = team.matches.create!(opponent_name: "Opp", scheduled_at: 1.hour.from_now)
session = StreamSession.create!(
match: match, user: user, platform: "matchlivetv", status: "idle", stream_node: node
)
expect(session.rtmp_ingest_url).to eq(
"rtmp://ingest-01.mltv-stream.net:1935/live/match_#{session.id}"
)
expect(session.hls_playback_url).to eq(
"https://ingest-01.mltv-stream.net/hls/live/match_#{session.id}/index.m3u8"
)
expect(session.mediamtx_api_base_url).to eq("http://10.0.0.2:9997")
expect(session.mediamtx_internal_rtmp_url).to eq("rtmp://10.0.0.2:1935")
end
end
@@ -1,7 +1,7 @@
require "rails_helper"
RSpec.describe Mediamtx::PublisherSync do
let(:user) { User.create!(email: "sync@test.com", name: "Sync", password: "password123", role: "coach") }
let(:user) { User.create!(email: "sync@test.com", name: "Sync", password: "Password123", role: "coach") }
let(:club) { Club.create!(name: "Sync Club", sport: "volleyball", primary_color: "#e53935", secondary_color: "#ffffff") }
let(:team) { club.teams.create!(name: "Under 16", sport: "volleyball") }
let!(:match) { team.matches.create!(opponent_name: "Avversario") }
@@ -20,8 +20,11 @@ RSpec.describe Mediamtx::PublisherSync do
before do
Billing::AssignPlan.call(club: club, plan_slug: "premium_full")
allow(Mediamtx::Client).to receive(:new).and_return(client)
allow(Mediamtx::Client).to receive(:for_session).and_return(client)
allow(Mediamtx::PublisherOnline).to receive(:path_info).and_return(path_info)
allow(Mediamtx::PublisherOnline).to receive(:active_path?).and_return(true)
allow(Mediamtx::PublisherOnline).to receive(:rtmp_publisher?).and_return(false)
allow(Mediamtx::PublisherOnline).to receive(:active?).and_return(true)
allow_any_instance_of(described_class).to receive(:redis).and_return(redis)
allow(client).to receive(:set_path_recording)
end
@@ -36,6 +39,7 @@ RSpec.describe Mediamtx::PublisherSync do
it "non abilita la registrazione in connecting senza publisher online" do
allow(Mediamtx::PublisherOnline).to receive(:active_path?).and_return(false)
allow(Mediamtx::PublisherOnline).to receive(:active?).and_return(false)
path_info["online"] = false
described_class.new(session).call
@@ -53,6 +57,7 @@ RSpec.describe Mediamtx::PublisherSync do
it "non disabilita la registrazione su reconnecting con publisher offline" do
session.update!(status: "reconnecting")
allow(Mediamtx::PublisherOnline).to receive(:active_path?).and_return(false)
allow(Mediamtx::PublisherOnline).to receive(:active?).and_return(false)
described_class.new(session).call
@@ -25,6 +25,7 @@ RSpec.describe Ops::HealthChecks do
%i[
check_recordings_size check_postgres check_redis check_mediamtx check_garage
check_sidekiq_heartbeat check_sidekiq_dead check_http_rails check_rails_latency
check_stream_overflow
].each do |method|
allow_any_instance_of(described_class).to receive(method).and_return(
described_class::Finding.new(
@@ -52,6 +53,7 @@ RSpec.describe Ops::HealthChecks do
%i[
check_recordings_size check_postgres check_redis check_mediamtx check_garage
check_sidekiq_heartbeat check_sidekiq_dead check_http_rails check_rails_latency
check_stream_overflow
].each do |method|
allow_any_instance_of(described_class).to receive(method).and_return(
described_class::Finding.new(
@@ -65,7 +67,7 @@ RSpec.describe Ops::HealthChecks do
summary = described_class.new.summary
expect(summary[:status]).to eq("ok")
expect(summary[:checks].size).to eq(10)
expect(summary[:checks].size).to eq(11)
end
it "degraded quando la latenza p95 supera la soglia warning" do
@@ -0,0 +1,178 @@
# frozen_string_literal: true
require "rails_helper"
RSpec.describe Streams::Autoscaler do
def with_env(vars)
previous = vars.keys.index_with { |k| ENV[k] }
vars.each { |k, v| ENV[k] = v }
yield
ensure
previous.each { |k, v| v.nil? ? ENV.delete(k) : ENV[k] = v }
end
let(:redis) { Redis.new(url: ENV.fetch("REDIS_URL", "redis://redis:6379/0")) }
before do
redis.del(Streams::Autoscaler::LOCK_KEY)
redis.del(Streams::Autoscaler::KILL_SWITCH_KEY)
redis.del(Streams::DnsProviders::Lab::REDIS_KEY)
StreamSession.where.not(stream_node_id: nil).update_all(stream_node_id: nil, status: "ended", ended_at: Time.current)
StreamNode.where.not(slug: Streams::NodeRegistry::HOME_SLUG).delete_all
end
it "is a no-op when disabled" do
with_env("STREAM_AUTOSCALE_ENABLED" => "0") do
result = described_class.reconcile!
expect(result.skipped).to eq(true)
expect(result.actions).to eq([])
end
end
it "scales out and creates a warm spare when free slots are low" do
with_env(
"STREAM_AUTOSCALE_ENABLED" => "1",
"STREAM_AUTOSCALE_SOFT_FREE_SLOTS" => "2",
"STREAM_AUTOSCALE_WARM_SPARE" => "1",
"STREAM_AUTOSCALE_KIND" => "lab",
"STREAM_AUTOSCALE_MAX_NODES" => "5",
"STREAM_CLOUD_PROVIDER" => "local_lab",
"STREAM_DNS_PROVIDER" => "lab",
"STREAM_NODE_HOME_MAX_PUBLISHERS" => "2",
"MEDIAMTX_API_URL" => "http://mtx-home:9997",
"MEDIAMTX_RTMP_URL" => "rtmp://home.example:1935",
"HLS_PUBLIC_URL" => "https://home.example/hls"
) do
home = Streams::NodeRegistry.ensure_home_from_env!
user = User.create!(email: "as@example.com", name: "A", password: "Password123", role: "coach")
club = Club.create!(name: "AS", sport: "volleyball")
team = club.teams.create!(name: "T", sport: "volleyball", slug: "as-t")
match = team.matches.create!(opponent_name: "X", scheduled_at: 1.hour.from_now)
# Fill home (2/2) → free_slots=0
2.times do
StreamSession.create!(match: match, user: user, platform: "matchlivetv", status: "live", stream_node: home)
end
provisioner = instance_double(Streams::NodeProvisioner)
created = []
allow(provisioner).to receive(:provision_lab!) do
node = StreamNode.create!(
slug: "ingest-lab-#{created.size + 1}",
hostname: "h#{created.size}.lab",
role: "lab",
status: "ready",
provider: "local",
rtmp_base_url: "rtmp://h:1935",
hls_base_url: "https://h/hls",
api_base_url: "http://h:9997",
max_publishers: 2,
max_relays: 2
)
created << node
node
end
result = described_class.reconcile!(provisioner: provisioner)
expect(result.actions).to include(:scale_out)
expect(created.size).to eq(1)
expect(result.metrics[:free_slots]).to be >= 2
# Riempi il nodo overflow → spare_ready=0 ma carico presente → warm spare
StreamSession.create!(
match: match, user: user, platform: "matchlivetv", status: "live", stream_node: created.first
)
StreamSession.create!(
match: match, user: user, platform: "matchlivetv", status: "live", stream_node: created.first
)
result2 = described_class.reconcile!(provisioner: provisioner)
expect(result2.actions).to include(:warm_spare).or include(:scale_out)
expect(created.size).to be >= 2
end
end
it "scales in an idle overflow node when warm spare is not required" do
with_env(
"STREAM_AUTOSCALE_ENABLED" => "1",
"STREAM_AUTOSCALE_SOFT_FREE_SLOTS" => "0",
"STREAM_AUTOSCALE_WARM_SPARE" => "0",
"STREAM_AUTOSCALE_IDLE_MINUTES" => "0",
"STREAM_NODE_HOME_MAX_PUBLISHERS" => "6",
"MEDIAMTX_API_URL" => "http://mtx-home:9997",
"MEDIAMTX_RTMP_URL" => "rtmp://home.example:1935",
"HLS_PUBLIC_URL" => "https://home.example/hls"
) do
Streams::NodeRegistry.ensure_home_from_env!
idle = StreamNode.create!(
slug: "ingest-lab-99",
hostname: "idle.lab",
role: "lab",
status: "ready",
provider: "local",
provider_instance_id: "sim-x",
rtmp_base_url: "rtmp://h:1935",
hls_base_url: "https://h/hls",
api_base_url: "http://h:9997",
max_publishers: 2,
max_relays: 2,
created_at: 2.hours.ago
)
provisioner = instance_double(Streams::NodeProvisioner)
allow(provisioner).to receive(:drain!)
expect(provisioner).to receive(:decommission!).with(idle)
result = described_class.reconcile!(provisioner: provisioner)
expect(result.actions.map(&:to_s)).to include("scale_in_ingest-lab-99")
end
end
it "skips when Redis kill-switch is engaged even if ENV is enabled" do
redis.del(described_class::KILL_SWITCH_KEY)
with_env("STREAM_AUTOSCALE_ENABLED" => "1") do
described_class.engage_kill_switch!
expect(described_class.enabled?).to eq(false)
result = described_class.reconcile!
expect(result.skipped).to eq(true)
ensure
described_class.clear_kill_switch!
end
end
it "blocks scale-out when next node would exceed monthly budget" do
with_env(
"STREAM_AUTOSCALE_ENABLED" => "1",
"STREAM_AUTOSCALE_SOFT_FREE_SLOTS" => "2",
"STREAM_AUTOSCALE_WARM_SPARE" => "0",
"STREAM_AUTOSCALE_KIND" => "lab",
"STREAM_AUTOSCALE_MAX_NODES" => "5",
"STREAM_AUTOSCALE_NODE_EUR_PER_HOUR" => "1",
"STREAM_AUTOSCALE_MONTHLY_BUDGET_EUR" => "10",
"STREAM_CLOUD_PROVIDER" => "local_lab",
"STREAM_DNS_PROVIDER" => "lab",
"STREAM_NODE_HOME_MAX_PUBLISHERS" => "1",
"MEDIAMTX_API_URL" => "http://mtx-home:9997",
"MEDIAMTX_RTMP_URL" => "rtmp://home.example:1935",
"HLS_PUBLIC_URL" => "https://home.example/hls"
) do
StreamSession.where.not(stream_node_id: nil).update_all(stream_node_id: nil, status: "ended", ended_at: Time.current)
StreamNode.where.not(slug: Streams::NodeRegistry::HOME_SLUG).delete_all
home = Streams::NodeRegistry.ensure_home_from_env!
home.update!(max_publishers: 1)
user = User.create!(email: "budget@example.com", name: "B", password: "Password123", role: "coach")
club = Club.create!(name: "Budget", sport: "volleyball")
team = club.teams.create!(name: "T", sport: "volleyball", slug: "budget-t")
match = team.matches.create!(opponent_name: "X", scheduled_at: 1.hour.from_now)
StreamSession.create!(match: match, user: user, platform: "matchlivetv", status: "live", stream_node: home)
provisioner = instance_double(Streams::NodeProvisioner)
expect(provisioner).not_to receive(:provision_lab!)
result = described_class.reconcile!(provisioner: provisioner)
expect(result.actions).to include(:blocked_capacity)
# 0 overflow nodes → stima 0€ ≤ budget; il blocco è sul nodo successivo (+1)
expect(described_class.within_budget?(0)).to eq(true)
expect(described_class.within_budget?(1)).to eq(false)
end
end
end
@@ -0,0 +1,82 @@
# frozen_string_literal: true
require "rails_helper"
RSpec.describe Streams::CloudProviders::Hetzner do
let(:stubs) { Faraday::Adapter::Test::Stubs.new }
let(:conn) do
Faraday.new(url: Streams::CloudProviders::Hetzner::API) do |f|
f.request :json
f.response :json
f.adapter :test, stubs
end
end
let(:provider) { described_class.new(token: "test-token", conn: conn) }
def with_env(vars)
previous = vars.keys.index_with { |k| ENV[k] }
vars.each { |k, v| ENV[k] = v }
yield
ensure
previous.each { |k, v| v.nil? ? ENV.delete(k) : ENV[k] = v }
end
after { stubs.verify_stubbed_calls }
it "creates a server and waits until running" do
with_env(
"HCLOUD_LOCATION" => "nbg1",
"HCLOUD_SERVER_TYPE" => "cpx12",
"HCLOUD_IMAGE" => "debian-12",
"HCLOUD_SSH_KEY" => "matchlivetv-stream-hetzner"
) do
stubs.post("https://api.hetzner.cloud/v1/servers") do |env|
body = JSON.parse(env.body)
expect(body["name"]).to eq("mltv-stream-ingest-01")
expect(body["location"]).to eq("nbg1")
[
201,
{ "Content-Type" => "application/json" },
{
"server" => {
"id" => 42,
"name" => "mltv-stream-ingest-01",
"status" => "initializing",
"public_net" => { "ipv4" => { "ip" => "49.13.1.2" } },
"private_net" => []
},
"action" => { "id" => 7, "status" => "running" }
}
]
end
stubs.get("https://api.hetzner.cloud/v1/actions/7") do
[200, { "Content-Type" => "application/json" }, { "action" => { "id" => 7, "status" => "success" } }]
end
stubs.get("https://api.hetzner.cloud/v1/servers/42") do
[
200,
{ "Content-Type" => "application/json" },
{
"server" => {
"id" => 42,
"name" => "mltv-stream-ingest-01",
"status" => "running",
"public_net" => { "ipv4" => { "ip" => "49.13.1.2" } },
"private_net" => []
}
}
]
end
instance = provider.create_node(name: "mltv-stream-ingest-01", labels: { role: "stream-node" })
expect(instance.id).to eq("42")
expect(instance.public_ip).to eq("49.13.1.2")
expect(instance.status).to eq("running")
end
end
it "destroys a server" do
stubs.delete("https://api.hetzner.cloud/v1/servers/42") { [204, {}, ""] }
expect(provider.destroy_node("42")).to eq(true)
end
end
@@ -0,0 +1,35 @@
# frozen_string_literal: true
require "rails_helper"
RSpec.describe Streams::DnsProviders::Hetzner do
let(:stubs) { Faraday::Adapter::Test::Stubs.new }
let(:conn) do
Faraday.new(url: Streams::DnsProviders::Hetzner::API) do |f|
f.request :json
f.response :json
f.adapter :test, stubs
end
end
let(:provider) { described_class.new(token: "test-token", zone: "mltv-stream.net", conn: conn) }
after { stubs.verify_stubbed_calls }
it "upserts an A record (delete then create)" do
stubs.delete("https://api.hetzner.cloud/v1/zones/mltv-stream.net/rrsets/ingest-01/A") { [404, {}, ""] }
stubs.post("https://api.hetzner.cloud/v1/zones/mltv-stream.net/rrsets") do |env|
body = JSON.parse(env.body)
expect(body["name"]).to eq("ingest-01")
expect(body["type"]).to eq("A")
expect(body["records"].first["value"]).to eq("49.13.1.2")
[201, { "Content-Type" => "application/json" }, { "rrset" => { "name" => "ingest-01" } }]
end
expect(provider.upsert_a("ingest-01.mltv-stream.net", "49.13.1.2")).to eq(true)
end
it "deletes an A record" do
stubs.delete("https://api.hetzner.cloud/v1/zones/mltv-stream.net/rrsets/ingest-01/A") { [204, {}, ""] }
expect(provider.delete_a("ingest-01.mltv-stream.net")).to eq(true)
end
end
@@ -0,0 +1,41 @@
# frozen_string_literal: true
require "rails_helper"
RSpec.describe "Streams::NodeProvisioner cloud" do
it "registers a cloud node using injected providers" do
cloud = instance_double(
Streams::CloudProviders::Hetzner,
create_node: Streams::CloudProviders::Instance.new(
id: "99",
name: "mltv-stream-ingest-01",
public_ip: "49.13.9.9",
private_ip: "10.0.0.9",
status: "running",
raw: {}
)
)
dns = instance_double(Streams::DnsProviders::Hetzner)
allow(dns).to receive(:upsert_a)
ENV["MEDIAMTX_API_URL"] = "http://mtx-home:9997"
ENV["MEDIAMTX_RTMP_URL"] = "rtmp://home.example:1935"
ENV["HLS_PUBLIC_URL"] = "https://home.example/hls"
ENV["STREAM_CLOUD_DNS_SUFFIX"] = "mltv-stream.net"
ENV["STREAM_CLOUD_MAX_PUBLISHERS"] = "4"
node = Streams::NodeProvisioner.new(cloud: cloud, dns: dns).provision_cloud!
expect(node.slug).to eq("ingest-01")
expect(node.role).to eq("cloud")
expect(node.provider).to eq("hetzner")
expect(node.hostname).to eq("ingest-01.mltv-stream.net")
expect(node.rtmp_base_url).to eq("rtmp://ingest-01.mltv-stream.net:1935")
expect(node.api_base_url).to eq("http://10.0.0.9:9997")
expect(dns).to have_received(:upsert_a).with("ingest-01.mltv-stream.net", "49.13.9.9")
ensure
%w[
MEDIAMTX_API_URL MEDIAMTX_RTMP_URL HLS_PUBLIC_URL
STREAM_CLOUD_DNS_SUFFIX STREAM_CLOUD_MAX_PUBLISHERS
].each { |k| ENV.delete(k) }
end
end
@@ -0,0 +1,66 @@
# frozen_string_literal: true
require "rails_helper"
RSpec.describe Streams::NodeProvisioner do
def with_env(vars)
previous = vars.keys.index_with { |k| ENV[k] }
vars.each { |k, v| ENV[k] = v }
yield
ensure
previous.each { |k, v| v.nil? ? ENV.delete(k) : ENV[k] = v }
end
let(:redis) { Redis.new(url: ENV.fetch("REDIS_URL", "redis://redis:6379/0")) }
before do
redis.del(Streams::DnsProviders::Lab::REDIS_KEY)
StreamSession.where.not(stream_node_id: nil).update_all(stream_node_id: nil, status: "ended", ended_at: Time.current)
StreamNode.where.not(slug: Streams::NodeRegistry::HOME_SLUG).delete_all
end
it "provisions and decommissions a simulated lab node" do
with_env(
"STREAM_CLOUD_PROVIDER" => "local_lab",
"STREAM_DNS_PROVIDER" => "lab",
"STREAM_LAB_DNS_SUFFIX" => "lab.mltv-stream.net",
"MEDIAMTX_API_URL" => "http://mtx-home:9997",
"MEDIAMTX_RTMP_URL" => "rtmp://ingest-home.example:1935",
"HLS_PUBLIC_URL" => "https://ingest-home.example/hls",
"STREAM_LAB_MAX_PUBLISHERS" => "2"
) do
provisioner = described_class.new
node = provisioner.provision_lab!
expect(node.slug).to eq("ingest-lab-01")
expect(node.role).to eq("lab")
expect(node.status).to eq("ready")
expect(node.hostname).to eq("ingest-lab-01.lab.mltv-stream.net")
expect(node.api_base_url).to eq("http://mtx-home:9997")
expect(Streams::DnsProviders::Lab.new.resolve(node.hostname)).to eq("127.0.0.1")
provisioner.decommission!(node)
expect(StreamNode.find_by(slug: "ingest-lab-01")).to be_nil
expect(Streams::DnsProviders::Lab.new.resolve("ingest-lab-01.lab.mltv-stream.net")).to be_nil
end
end
it "refuses to decommission a busy node" do
with_env(
"STREAM_CLOUD_PROVIDER" => "local_lab",
"STREAM_DNS_PROVIDER" => "lab",
"MEDIAMTX_API_URL" => "http://mtx-home:9997",
"MEDIAMTX_RTMP_URL" => "rtmp://ingest-home.example:1935",
"HLS_PUBLIC_URL" => "https://ingest-home.example/hls"
) do
node = described_class.new.provision_lab!
user = User.create!(email: "lab@example.com", name: "L", password: "Password123", role: "coach")
club = Club.create!(name: "LabClub", sport: "volleyball")
team = club.teams.create!(name: "T", sport: "volleyball", slug: "lab-t")
match = team.matches.create!(opponent_name: "X", scheduled_at: 1.hour.from_now)
StreamSession.create!(match: match, user: user, platform: "matchlivetv", status: "live", stream_node: node)
expect { described_class.new.decommission!(node) }.to raise_error(Streams::NodeProvisioner::BusyError)
end
end
end
@@ -0,0 +1,95 @@
# frozen_string_literal: true
require "rails_helper"
RSpec.describe Streams::NodeRegistry do
def with_env(vars)
previous = vars.keys.index_with { |k| ENV[k] }
vars.each { |k, v| ENV[k] = v }
yield
ensure
previous.each { |k, v| v.nil? ? ENV.delete(k) : ENV[k] = v }
end
let(:home_env) do
{
"MEDIAMTX_API_URL" => "http://mtx-home:9997",
"MEDIAMTX_RTMP_URL" => "rtmp://ingest-home.example:1935",
"HLS_PUBLIC_URL" => "https://ingest-home.example/hls",
"MEDIAMTX_INTERNAL_RTMP_URL" => "rtmp://mtx-home:1935",
"MEDIAMTX_HLS_URL" => "http://mtx-home:8888",
"STREAM_NODE_HOME_MAX_PUBLISHERS" => "2"
}
end
before do
StreamSession.where.not(stream_node_id: nil).update_all(stream_node_id: nil, status: "ended", ended_at: Time.current)
StreamNode.where.not(slug: Streams::NodeRegistry::HOME_SLUG).delete_all
StreamNode.find_by(slug: Streams::NodeRegistry::HOME_SLUG)&.update!(max_publishers: 2, status: "ready")
end
it "creates home node from ENV" do
with_env(home_env) do
node = described_class.ensure_home_from_env!
expect(node.slug).to eq("home")
expect(node.rtmp_base_url).to eq("rtmp://ingest-home.example:1935")
expect(node.api_base_url).to eq("http://mtx-home:9997")
expect(node.max_publishers).to eq(2)
expect(node.status).to eq("ready")
end
end
it "does not reset home drain/offline/error status on ensure" do
with_env(home_env) do
home = described_class.ensure_home_from_env!
home.update!(status: "draining")
expect(described_class.ensure_home_from_env!.status).to eq("draining")
expect { described_class.allocate! }.to raise_error(Streams::NodeRegistry::NoCapacityError)
end
end
it "allocates the least loaded ready node" do
with_env(home_env) do
home = described_class.ensure_home_from_env!
cloud = StreamNode.create!(
slug: "ingest-01",
hostname: "ingest-01.mltv-stream.net",
role: "cloud",
status: "ready",
provider: "hetzner",
rtmp_base_url: "rtmp://ingest-01.mltv-stream.net:1935",
hls_base_url: "https://ingest-01.mltv-stream.net/hls",
api_base_url: "http://10.0.0.2:9997",
max_publishers: 2,
max_relays: 2
)
user = User.create!(email: "n@example.com", name: "N", password: "Password123", role: "coach")
club = Club.create!(name: "C", sport: "volleyball")
team = club.teams.create!(name: "T", sport: "volleyball", slug: "t-node")
match = team.matches.create!(opponent_name: "X", scheduled_at: 1.hour.from_now)
StreamSession.create!(
match: match, user: user, platform: "matchlivetv", status: "live", stream_node: home
)
expect(described_class.allocate!).to eq(cloud)
end
end
it "raises when no capacity remains" do
with_env(home_env.merge("STREAM_NODE_HOME_MAX_PUBLISHERS" => "1")) do
home = described_class.ensure_home_from_env!
user = User.create!(email: "full@example.com", name: "F", password: "Password123", role: "coach")
club = Club.create!(name: "Full", sport: "volleyball")
team = club.teams.create!(name: "T", sport: "volleyball", slug: "t-full")
match = team.matches.create!(opponent_name: "X", scheduled_at: 1.hour.from_now)
StreamSession.create!(
match: match, user: user, platform: "matchlivetv", status: "live", stream_node: home
)
expect { described_class.allocate! }.to raise_error(Streams::NodeRegistry::NoCapacityError)
end
end
end
@@ -21,4 +21,46 @@ RSpec.describe Streams::YoutubeRelay do
expect(cmd).not_to include("-b:a")
end
end
describe "multi-host sticky stop/capacity" do
let(:redis) { Redis.new(url: ENV.fetch("REDIS_URL", "redis://redis:6379/0")) }
let(:session_id) { SecureRandom.uuid }
before do
allow(described_class).to receive(:worker?).and_return(true)
allow(described_class).to receive(:worker_id).and_return("worker-a")
allow(described_class).to receive(:max_concurrent).and_return(1)
redis.flushdb
end
def session_double
instance_double(
StreamSession,
id: session_id,
platform: "youtube",
terminal?: false,
stream_key: "yt-key",
status: "live"
)
end
it "does not stop ffmpeg owned by another worker" do
redis.set(format(Streams::YoutubeRelay::OWNER_KEY, session_id), "worker-b", ex: 3600)
redis.set(format(Streams::YoutubeRelay::REDIS_KEY, session_id), "12345", ex: 3600)
expect(described_class.stop_on_worker!(session_double)).to eq(:wrong_host)
expect(redis.get(format(Streams::YoutubeRelay::OWNER_KEY, session_id))).to eq("worker-b")
end
it "requeues ensure when at capacity" do
other_id = SecureRandom.uuid
redis.sadd(format(Streams::YoutubeRelay::OWNED_SET, "worker-a"), other_id)
allow(described_class).to receive(:intake_available?).and_return(true)
expect(YoutubeRelayEnsureJob).to receive(:set).with(hash_including(wait: 5.seconds, queue: :youtube_relay)).and_return(
double(perform_later: true)
)
expect(described_class.ensure_on_worker!(session_double)).to eq(:at_capacity)
end
end
end
+5 -3
View File
@@ -2,8 +2,8 @@
Documento di riferimento per l'intero sistema: rete, container Docker, flussi video, replay, monitoraggio ops e rilasci.
**Ultimo aggiornamento:** 2026-07-29
**Produzione:** `eminux@192.168.1.146``/opt/matchlivetv`
**Ultimo aggiornamento:** 2026-08-09
**Produzione:** `eminux@192.168.1.146``/opt/matchlivetv` (Proxmox; control plane attuale per autoscale)
---
@@ -397,7 +397,8 @@ cd infra && cp .env.example .env && docker compose up -d --build
| Documento | Contenuto |
|-----------|-----------|
| [`infrastructure/SERVER_DEPLOYMENT.md`](infrastructure/SERVER_DEPLOYMENT.md) | Bootstrap server, NPM, cron, backup |
| [`infrastructure/HETZNER_CLOUD_RELAY_OVERFLOW.md`](infrastructure/HETZNER_CLOUD_RELAY_OVERFLOW.md) | Piano overflow relay YouTube su Hetzner Cloud (picchi) |
| [`infrastructure/STREAMING_AUTOSCALE.md`](infrastructure/STREAMING_AUTOSCALE.md) | **Design attuale:** Proxmox prod + Hetzner Cloud warm spare (MediaMTX+ffmpeg), lab, DNS, disco; Auction in futuro |
| [`infrastructure/HETZNER_CLOUD_RELAY_OVERFLOW.md`](infrastructure/HETZNER_CLOUD_RELAY_OVERFLOW.md) | Piano storico overflow-only relay YouTube (superseded dal doc sopra) |
| [`ANDROID_APP_LINKS.md`](ANDROID_APP_LINKS.md) | Digital Asset Links / deep link Play (`assetlinks.json`) |
| [`LIVE_STREAMING.md`](LIVE_STREAMING.md) | MediaMTX, pausa, HLS, ruolo ffmpeg |
| [`REPLAY_MODULE.md`](REPLAY_MODULE.md) | Garage, retention, YouTube VOD |
@@ -420,3 +421,4 @@ cd infra && cp .env.example .env && docker compose up -d --build
| 2026-06 | Cleanup path MediaMTX orfani (`Mediamtx::CleanupOrphanPaths`); delete path sempre a fine diretta |
| 2026-07 | Rimosso overlay server (`OverlayRelay`); tabellone bruciato in app; HLS su path camera (non `*_air`); `YoutubeRelay` = copy video + AAC |
| 2026-08 | `YoutubeRelay` = remux copy video+audio (niente ricodifica AAC); profilo app/slate AAC 48k mono |
| 2026-08 | Design autoscale: Proxmox prod + Hetzner Cloud warm spare (MediaMTX+ffmpeg), lab locale, secondo dominio DNS, Garage ora / B2 dopo; Auction rimandato (`STREAMING_AUTOSCALE.md`) |
@@ -1,10 +1,12 @@
# Piano: overflow relay YouTube su Hetzner Cloud
**Stato:** piano / design — non implementato
**Ultimo aggiornamento:** 2026-07-29
> **Superseded (2026-08-09):** il design corrente è [`STREAMING_AUTOSCALE.md`](STREAMING_AUTOSCALE.md) (Auction + Proxmox, scale di MediaMTX **e** ffmpeg, warm spare, secondo dominio DNS, Garage ora / B2 dopo). Questo file resta come riferimento storico sul solo overflow dei relay.
**Stato:** piano / design storico — non implementato come doc standalone
**Ultimo aggiornamento:** 2026-07-29 (header superseded 2026-08-09)
**Contesto:** produzione attuale su host dedicato (es. Proxmox/Hetzner) con MediaMTX + Sidekiq sullo stesso box; obiettivo >10 live YouTube contemporanee senza abbandonare il middle-tier.
Documenti correlati: [`LIVE_STREAMING.md`](../LIVE_STREAMING.md), [`ARCHITECTURE.md`](../ARCHITECTURE.md), [`OPS_MONITORING.md`](../OPS_MONITORING.md).
Documenti correlati: [`STREAMING_AUTOSCALE.md`](STREAMING_AUTOSCALE.md), [`LIVE_STREAMING.md`](../LIVE_STREAMING.md), [`ARCHITECTURE.md`](../ARCHITECTURE.md), [`OPS_MONITORING.md`](../OPS_MONITORING.md).
---
+372
View File
@@ -0,0 +1,372 @@
# Autoscale streaming — Proxmox attuale + Hetzner Cloud
**Stato:** fasi 04 implementate sul branch — **solo test lab; nessun deploy produzione**
**Branch:** `feature/streaming-autoscale-hetzner`
**Ultimo aggiornamento:** 2026-08-09 (Fase 4 hardening + runbook)
**Sostituisce / estende:** il piano overflow-only [`HETZNER_CLOUD_RELAY_OVERFLOW.md`](HETZNER_CLOUD_RELAY_OVERFLOW.md). Questo documento copre **MediaMTX + ffmpeg**, warm spare, DNS, storage, disco e lab sul Proxmox di produzione.
Documenti correlati: [`ARCHITECTURE.md`](../ARCHITECTURE.md), [`LIVE_STREAMING.md`](../LIVE_STREAMING.md), [`REPLAY_MODULE.md`](../REPLAY_MODULE.md), [`OPS_MONITORING.md`](../OPS_MONITORING.md).
---
## 1. Obiettivo
Scalare molte dirette contemporanee (sito e/o YouTube) senza saturare un unico box:
- **Control plane (ora):** Proxmox **già in produzione** (`192.168.1.146` / `/opt/matchlivetv`) — sito, API, DB, Redis, Garage, MediaMTX/ffmpeg “home”.
- **Data plane elastico:** **Hetzner Cloud** — nodi `MediaMTX + ffmpeg` con warm spare (+1).
- **DNS ingest:** secondo dominio su **Hetzner DNS** (API).
- **Lab:** stesso Proxmox, con `ProxmoxLabProvider`, prima di affidarsi al Cloud in produzione.
- **Futuro (non bloccante):** migrazione control plane su **Hetzner Auction + Proxmox** (vSwitch nativo, hardware dedicato).
---
## 2. Decisioni chiuse
| # | Decisione | Scelta |
|---|----------|--------|
| D1 | Host primario **ora** | **Proxmox già usato in produzione** (non attendere Auction) |
| D1b | Host primario **futuro** | Hetzner Server Auction + Proxmox (quando si migra) |
| D2 | Overflow / warm spare | **Hetzner Cloud** (attivato per questo progetto) |
| D3 | Unità di scala | Nodo **stream** = MediaMTX + worker `youtube_relay` (ffmpeg remux copy) |
| D4 | Warm pool | Almeno **1 spare ready** quando il carico si avvicina al cap |
| D5 | DNS ingest | Dominio **`mltv-stream.net`** (registrar Aruba, NS → Hetzner DNS). `matchlivetv.it` resta su Aruba intatto |
| D6 | Aruba | Nessuna delega NS; niente Floating IP prenotate |
| D7a | Rete **ora** (casa/Proxmox ↔ Cloud) | **WireGuard** (o Tailscale) privato: Redis, DB, API MediaMTX home. Non esporre Redis/Postgres su WAN |
| D7b | Rete **futuro** (Auction ↔ Cloud) | vSwitch Robot + Cloud Network (stessa location) |
| D8 | Object storage | **Garage resta**. **B2** = migrazione pronta quando misurato |
| D9 | Staging | Segmenti su disco VM `stream-*` in live; upload Garage a fine sessione |
| D10 | Prodotto | Consigliare **YouTube** per alleggerire storage/egress replay |
| D11 | Disco | Separazione volumi + **monitoraggio attivo** obbligatorio (§6) |
| D12 | Lab prima del Cloud prod | `ProxmoxLabProvider` + DNS lab sullo stesso Proxmox; poi `HetznerCloudProvider` |
---
## 3. Topologia — fase attuale (Proxmox prod + Cloud)
```text
Internet
│ HTTPS :443 · RTMP :1935
┌──────────────────────────────────────────────────────────┐
│ Proxmox produzione (attuale) │
│ eminux@192.168.1.146 · /opt/matchlivetv │
│ │
│ edge · rails · sidekiq-core · postgres · redis │
│ garage · stream-home (MediaMTX + youtube_relay) │
│ autoscaler (Rails/Sidekiq) │
│ │
│ Lab (opz.): clone VM stream-* via ProxmoxLabProvider │
└────────────────────────┬─────────────────────────────────┘
│ WireGuard / tunnel privato
┌──────────────────────────────────────────────────────────┐
│ Hetzner Cloud — pool stream nodes │
│ stream-01 … stream-N MediaMTX + ffmpeg │
│ + 1 spare ready (warm) │
└────────────────────────┬─────────────────────────────────┘
Hetzner DNS (secondo dominio)
ingest-XX.<dominio> → IP pubblico nodo
```
**Futuro Auction:** stesso schema, tunnel sostituito da vSwitch; control plane migrato sul bare metal Hetzner.
**Principio:** Rails solo controllo. Live sui MediaMTX del nodo assegnato. Replay su Garage (poi eventualmente B2).
---
## 4. DNS (secondo dominio)
RTMP **non** tollera round-robin su un unico hostname.
1. Dominio **`mltv-stream.net`** registrato su Aruba; zona DNS su Hetzner Console (progetto `matchlivetv-stream`).
2. NS Aruba del dominio → `hydrogen` / `oxygen` / `helium` Hetzner (delegazione OK).
3. Autoscaler a VM ready → upsert `A` `ingest-03.mltv-stream.net` → health-check → nodo `ready`.
4. API create/start sessione → URL del nodo assegnato:
```text
rtmp://ingest-03.mltv-stream.net:1935/live/match_<uuid>
https://ingest-03.mltv-stream.net/hls/... # o via edge matchlivetv.it che proxya il nodo
```
5. Scale-in: drain → delete DNS → destroy VM.
Hostname pubblici dei nodi vivono sotto `*.mltv-stream.net`. In lab: `LabDnsProvider` (`/etc/hosts`, dnsmasq).
---
## 5. Autoscaler e warm spare
Ogni nodo: `max_publishers` / `max_relays` (es. 4).
| Regola | Comportamento |
|--------|----------------|
| Scale-out | `free_slots` sotto soglia soft → crea 1 nodo |
| Warm spare | Se `spare_ready < 1` vicino al carico → crea spare |
| Scale-in | Idle da X min e non unica spare → destroy |
| Floor | `stream-home` sul Proxmox sempre on |
`CloudProvider`: `HetznerCloudProvider` (prod overflow) + `ProxmoxLabProvider` (test).
Code Sidekiq: coda `youtube_relay` isolata; sticky owner Redis.
---
## 6. Disco Proxmox — requisito critico
**Non** mettere OS, DB e video sullo stesso volume che può riempirsi al 100%.
Vale **già** sul Proxmox di produzione (rinforzare ora), e di nuovo alla migrazione Auction.
### 6.1 Layout volumi (target)
| Volume / disco | Contenuto | Note |
|----------------|-----------|------|
| `os` | Proxmox / sistema host | Quasi statico |
| `vm-system` / root stack | rails, edge, redis, … | Snapshot |
| `vm-data-db` | Postgres | Separato; backup prioritari |
| `vm-data-video` | Garage + staging MediaMTX | **Disco che può riempirsi** |
| `backup` | Backup / PBS | **Mai** sullo stesso disco dei video |
Regole: cap Garage/staging; se staging pieno → **rifiutare nuove sessioni**, non far cadere DB/API.
### 6.2 Monitoraggio attivo (obbligatorio)
| Metrica | Warn | Critical |
|---------|------|----------|
| Uso `%` volume video | ≥ 70% | ≥ 85% |
| Spazio libero GB | sotto soglia | &lt; X GB |
| Inode liberi | bassi | esaurimento |
| Crescita GB/giorno Garage | anomalia | — |
| Purge/retention falliti | — | immediato |
| Staging per nodo `stream-*` | ≥ 70% | ≥ 85% + blocco live su quel nodo |
Checklist:
- [ ] Alert ntfy disco testati
- [ ] Vista ops “disco video”
- [ ] Test periodico “80% pieno”
- [ ] Restore di prova Postgres
---
## 7. Storage: Garage ora, B2 dopo
| Ora | **Garage** |
| Dopo | **Backblaze B2** quando metriche/€ lo giustificano |
Client S3-ready; staging locale invariato. Spinta prodotto YouTube riduce pressione replay.
---
## 8. Prodotto: YouTube consigliato
Suggerire YouTube in app/dashboard quando ha senso → meno storage/egress MatchLiveTV; middle-tier (slate + relay) resta il valore. Live solo-sito restano supportate.
---
## 9. Astrazioni codice
```text
CloudProvider
create_node / destroy_node / list_nodes / wait_until_running / public_ip
→ HetznerCloudProvider | ProxmoxLabProvider
DnsProvider
upsert_a / delete_a
→ HetznerDnsProvider | LabDnsProvider
StreamNodeRegistry
register / mark_ready / allocate_for_session / drain / release
Autoscaler
reconcile!(metrics) → ensure spare / scale-in idle
```
Tag VM Cloud: `matchlivetv`, `role=stream-node`, `env=prod|lab`.
---
## 10. Lab sul Proxmox attuale
Prima (o in parallelo) al Cloud “vero”:
1. Template VM `stream-node` su Proxmox.
2. `CLOUD_PROVIDER=proxmox_lab` → clone/start/stop via API Proxmox.
3. DNS lab (hosts/dnsmasq).
4. Soft cap basso; N sessioni di test; verifica assignment, spare, scale-in, alert disco.
5. Poi smoke su Hetzner Cloud (1 CX piccolo) + DNS reale.
Cosa il lab **non** replica al 100%: tempi boot Hetzner, vSwitch, RTMP 4G multi-nodo (smoke WAN dopo).
---
## 11. Fasi di implementazione
| Fase | Cosa | Esito |
|------|------|--------|
| **L — Lab Proxmox** | `LocalLab` / `ProxmoxLab`, DNS lab, admin nodi | **implementata sul branch** |
| **0 — Multi-nodo ready** | Registry + URL RTMP/HLS in API (`StreamNode`, `Streams::NodeRegistry`) | **implementata sul branch** |
| **1 — Hetzner Cloud + DNS** | `HetznerCloudProvider` + `HetznerDnsProvider`, `mltv-stream.net`, cloud-init, WireGuard (§16) | **implementata sul branch** (WG ops manuale) |
| **2 — Routing relay** | Coda `youtube_relay`, sticky owner, cap `RELAY_MAX_CONCURRENT` | **implementata sul branch** |
| **3 — Autoscaler** | Soglie + warm spare (+1), kill-switch `STREAM_AUTOSCALE_ENABLED` | **implementata sul branch** |
| **4 — Hardening** | Drain sicuro, budget, kill-switch Redis/admin, alert overflow, runbook | **implementata sul branch** (solo test lab — **non in prod**) |
| **A — Auction (futuro)** | Migrazione control plane + vSwitch al posto di WireGuard | Hardware dedicato Hetzner |
| **B2 (opz.)** | Switch object storage | Quando misurato |
### Fase 0 — dettagli implementati
- Tabella `stream_nodes` + `stream_sessions.stream_node_id`
- `Streams::NodeRegistry.ensure_home_from_env!` / `allocate!` (least-loaded)
- `Sessions::Create` assegna il nodo e crea il path MediaMTX sul client del nodo
- URL RTMP/HLS/API/internal per sessione derivati dal nodo (fallback ENV se nodo assente)
- API JSON espone `stream_node` (slug)
- Dominio ingest pubblico futuro: `*.mltv-stream.net`
### Fase L — dettagli implementati
- `Streams::CloudProviders` (`local_lab`, `proxmox_lab`, stub `hetzner`)
- `Streams::DnsProviders` (`lab` su Redis, stub `hetzner`)
- `Streams::NodeProvisioner` (provision / drain / decommission)
- Admin **Nodi stream** (`/admin/stream_nodes`): lista, provision lab, drain, delete
- Rake: `streams:nodes:ensure_home`, `streams:nodes:provision_lab`, `streams:nodes:dns_lab_dump`
- Default sicuro: `STREAM_CLOUD_PROVIDER=local_lab` (nessuna chiamata Proxmox finché non configuri token)
### Fase 1 — dettagli implementati
- `Streams::CloudProviders::Hetzner` (create/destroy server, wait running, labels)
- `Streams::DnsProviders::Hetzner` (upsert/delete A su zona `mltv-stream.net`)
- `NodeProvisioner#provision_cloud!` + bottone admin **Provisiona nodo Hetzner**
- Cloud-init: `infra/stream-node/cloud-init.yaml` (`HCLOUD_USER_DATA_FILE`)
- Decommission usa il provider del nodo (non solo ENV globale)
- WireGuard: vedi §16
### Fase 2 — dettagli implementati
- Coda Sidekiq dedicata `youtube_relay` (priorità sopra `default`)
- `YoutubeRelayEnsureJob` / `YoutubeRelayStopJob` solo su quella coda
- Stop sticky: non cancella owner da Rails; lo stop gira sullowner o requeue (`:wrong_host`)
- Cap per worker: `RELAY_MAX_CONCURRENT` (default 4) + set Redis `youtube_relay:owned:HOSTNAME`
- Ensure a capacità piena → requeue 5s invece di avviare un secondo ffmpeg locale
### Fase 3 — dettagli implementati
- `Streams::Autoscaler` + job chain `Streams::AutoscalerJob` (come health monitor)
- Kill-switch: `STREAM_AUTOSCALE_ENABLED=1` per attivare
- Scale-out se `free_slots ≤ STREAM_AUTOSCALE_SOFT_FREE_SLOTS`
- Warm spare (+N) se carico/soglia e `spare_ready < WARM_SPARE`
- Scale-in overflow idle da `IDLE_MINUTES` (non tocca home; rispetta warm spare)
- Cap `STREAM_AUTOSCALE_MAX_NODES`; kind `lab|cloud`
- Metriche in admin Nodi stream
### Fase 4 — hardening (implementata sul branch)
> **NON rilasciare in produzione** finché lab + smoke Cloud non sono verdi. Su questo branch restano `STREAM_AUTOSCALE_ENABLED=0` e `STREAM_AUTOSCALE_ALLOW_CLOUD=0`.
- Scale-in **sicuro**: `drain!` → attesa sessioni zero → `decommission!`
- Budget soft: `STREAM_AUTOSCALE_MONTHLY_BUDGET_EUR` + `STREAM_AUTOSCALE_NODE_EUR_PER_HOUR` (stima 24/7); scale-out bloccato se sforerebbe
- Cloud autoscale solo con `STREAM_AUTOSCALE_ALLOW_CLOUD=1` **e** `HCLOUD_TOKEN`
- Kill-switch runtime Redis `streams:autoscaler:kill_switch` (bottoni admin ON/OFF) oltre allENV
- Health check `stream_overflow` → incidente Ops/ntfy su nodi idle orfani, over-budget, capacità al max
- Admin: metriche budget + stato kill-switch
#### Runbook operativo (lab / pre-prod)
| Situazione | Azione |
|------------|--------|
| Dubbio / picco anomalo | Admin → **Kill-switch ON** (o `STREAM_AUTOSCALE_ENABLED=0`) |
| Nodo spillato | Drain dal admin; dopo fine live → Destroy |
| Budget alert Ops | Abbassa `MAX_NODES` / alza budget solo dopo review costi Hetzner |
| Smoke Cloud | Provision **manuale** admin (kind cloud), non autoscaler; verifica WG + RTMP 1935 |
| Scale-out lab OK | Solo dopo: considerare `ENABLED=1` + `KIND=lab` su staging/lab Proxmox |
| Produzione overflow | Solo dopo checklist §12 punto 6 |
Ordine di lavoro consigliato sul branch: **0 → L → 1 → 2 → 3 → 4**; **A** quando si decide di lasciare il Proxmox casa.
---
## 12. Ordine operativo “ora” (senza Auction)
1. Rinforzare dischi/alert sul Proxmox prod (§6).
2. Secondo dominio + zona Hetzner DNS.
3. Progetto Hetzner Cloud + immagine/snapshot `stream-node`.
4. Tunnel WireGuard Proxmox ↔ Cloud (Redis/DB/API private).
5. Implementare fasi 0 → L → 1… sul branch.
6. Cutover overflow in produzione solo dopo lab + smoke Cloud verdi.
---
## 13. Criteri di successo
| Criterio | Misura |
|----------|--------|
| Lab | Scale-out/in e assignment verificati su Proxmox senza Hetzner |
| Picco | N live oltre soft cap home con spare Cloud ready |
| Continuità | Drop telefono → slate YT attiva |
| Disco | Nessun outage DB/API per disco video; alert prima del critical |
| DNS | Aruba intatto; ingest sul secondo dominio |
| Futuro | Path chiaro verso Auction senza riscrivere autoscaler |
---
## 14. Aperto / da calibrare
| Voce | Note |
|------|------|
| Nome secondo dominio | **`mltv-stream.net`** (chiuso) |
| Soft cap `stream-home` | es. 4 vs 6 |
| Tipo VM Cloud | CPX vs CCX |
| `OVERFLOW_MAX` / budget | — |
| Spare di notte | 0 vs 1 |
| Dettaglio WireGuard | subnet, peer, MTU |
| Soglie GB disco | da size reale volume video |
---
## 15. Riepilogo esecutivo
**Ora:** restiamo sul Proxmox di produzione; attiviamo Hetzner Cloud (spare MediaMTX+ffmpeg) + Hetzner DNS (secondo dominio) + WireGuard. Testiamo prima in lab sullo stesso Proxmox. Garage resta; B2 dopo. YouTube consigliato in prodotto. Disco e alert sono vincolo non negoziabile.
**Dopo:** eventuale Auction Hetzner sostituisce il control plane casa e WireGuard → vSwitch, senza cambiare il modello nodi/autoscaler.
---
## 16. WireGuard (Proxmox casa ↔ Hetzner Cloud)
Obiettivo: Rails/Sidekiq sullhost di produzione raggiungono `api_base_url` / `internal_rtmp_url` dei nodi Cloud (es. `http://10.0.0.9:9997`) **senza** esporre Redis/Postgres/MediaMTX API su Internet.
### Setup consigliato (una tantum)
1. Su Hetzner Console: crea **Network** privata (es. `10.0.0.0/16`) nella location `fsn1`, subnet `10.0.1.0/24` per i server Cloud.
2. Imposta `HCLOUD_NETWORK_ID=<id>` così i nodi stream entrano nella network al create.
3. Su Proxmox (VM o LXC gateway): installa WireGuard; peer verso una VM “gateway” Cloud **sempre on** (CX22 piccola) *oppure* peer site-to-site verso un server Cloud fisso.
4. Alternativa più semplice per i primi test: **Tailscale** su Proxmox host + su ogni stream-node (cloud-init) — stesso effetto, meno ops.
5. Firewall Hetzner Cloud:
- WAN: `1935/tcp` (RTMP), `443/tcp` o `8888` se HLS diretto
- Solo rete privata / WG: `9997` (MediaMTX API), niente Postgres/Redis sui nodi stream
6. Verifica da Rails: `curl http://<private_ip>:9997/v3/paths/list`
Finché WireGuard/Tailscale non è pronto, `api_base_url` punta comunque alla private IP (o pubblica se manca private_net): il path MediaMTX da Create fallirà dal control plane se non raggiungibile — provisionare nodi Cloud solo dopo connettività privata OK, oppure smoke con API su IP pubblico temporaneo + firewall allowlist IP casa.
### ENV produzione (stream autoscale)
```bash
HCLOUD_TOKEN=...
HCLOUD_LOCATION=nbg1
HCLOUD_SERVER_TYPE=cpx12
HCLOUD_IMAGE=debian-12
HCLOUD_SSH_KEY=matchlivetv-stream-hetzner
HCLOUD_NETWORK_ID= # opzionale
HCLOUD_USER_DATA_FILE=/opt/matchlivetv/infra/stream-node/cloud-init.yaml
STREAM_DNS_ZONE=mltv-stream.net
STREAM_DNS_TTL=60
STREAM_CLOUD_DNS_SUFFIX=mltv-stream.net
STREAM_CLOUD_MAX_PUBLISHERS=4
STREAM_NODE_ENV=prod
# Default lab sicuro; in prod per overflow usa hetzner esplicitamente dal bottone admin
STREAM_CLOUD_PROVIDER=local_lab
STREAM_DNS_PROVIDER=lab
```
+109
View File
@@ -0,0 +1,109 @@
# Copia come infra/.env sul container di collaudo e genera i secret.
# cp .env.collaudo.example .env && openssl rand -hex 32 (ripeti per ogni CHANGE_ME)
POSTGRES_PASSWORD=CHANGE_ME_STRONG_PASSWORD
SECRET_KEY_BASE=CHANGE_ME_openssl_rand_hex_64
JWT_SECRET=CHANGE_ME_openssl_rand_hex_32
MEDIAMTX_WEBHOOK_SECRET=CHANGE_ME_openssl_rand_hex_32
# RTMP collaudo: porta host 11935 (vedi docker-compose.collaudo.yml).
# LAN: rtmp://192.168.1.157:11935
# WAN (opzionale): apri SOLO 11935 sul router → .157 — NON toccare :1935 di produzione.
MEDIAMTX_RTMP_URL=rtmp://192.168.1.157:11935
APP_PUBLIC_URL=https://collaudo.matchlivetv.it
HLS_PUBLIC_URL=https://collaudo.matchlivetv.it/hls
CORS_ORIGINS=https://collaudo.matchlivetv.it
ALLOWED_HOSTS=collaudo.matchlivetv.it,192.168.1.157,localhost
YOUTUBE_REDIRECT_URI=https://collaudo.matchlivetv.it/api/v1/youtube/callback
YOUTUBE_CLIENT_ID=
YOUTUBE_CLIENT_SECRET=
YOUTUBE_PLATFORM_REFRESH_TOKEN=
# Stripe: preferisci chiavi Test su collaudo
STRIPE_SECRET_KEY=
STRIPE_WEBHOOK_SECRET=
STRIPE_PREMIUM_LIGHT_MONTHLY_PRICE_ID=
STRIPE_PREMIUM_LIGHT_YEARLY_PRICE_ID=
STRIPE_PREMIUM_FULL_MONTHLY_PRICE_ID=
STRIPE_PREMIUM_FULL_YEARLY_PRICE_ID=
STRIPE_PREMIUM_LIGHT_PRICE_ID=
STRIPE_PREMIUM_FULL_PRICE_ID=
RAILS_LOG_LEVEL=info
PRIVACY_CONTROLLER_NAME=Emiliano Frascaro
PRIVACY_CONTROLLER_ADDRESS=Via Guido De Ruggiero, 89 - 20142 - Milano (MI)
PRIVACY_CONTACT_EMAIL=privacy@matchlivetv.it
PRIVACY_CONTROLLER_VAT=
MAILER_FROM=Match Live TV Collaudo <noreply@matchlivetv.it>
SMTP_ADDRESS=
SMTP_PORT=465
SMTP_USERNAME=
SMTP_PASSWORD=
SMTP_AUTH=plain
SMTP_SSL=true
SMTP_STARTTLS=false
PASSWORD_RESET_EXPIRY_HOURS=2
MATCHLIVETV_VIDEOS_ROOT=/media/videos/matchlivetv
REPLAY_STORAGE_ENDPOINT=http://garage:3900
REPLAY_STORAGE_BUCKET=matchlivetv-replays
REPLAY_STORAGE_REGION=garage
REPLAY_STORAGE_ACCESS_KEY_ID=
REPLAY_STORAGE_SECRET_ACCESS_KEY=
REPLAY_STORAGE_FORCE_PATH_STYLE=true
REPLAY_MEDIA_PUBLIC_BASE_URL=https://collaudo.matchlivetv.it/media
REPLAY_MEDIA_REDIRECT=true
RAILS_MAX_THREADS=5
OPS_HTTP_RAILS_URL=http://edge/up
OPS_HTTP_PUBLIC_INTERVAL_SECS=900
NTFY_PUBLIC_URL=http://192.168.1.157:18090
OPS_NTFY_URL=
OPS_NTFY_TOKEN=
OPS_ALERT_EMAIL=
OPS_NOTIFY_SEVERITIES=critical,warning
OPS_NOTIFY_COOLDOWN_MINUTES=30
OPS_HEALTH_INTERVAL_SECS=180
OPS_HEALTH_TOKEN=
OPS_DISK_WARN_PERCENT=80
OPS_DISK_CRIT_PERCENT=90
OPS_RECORDINGS_WARN_GB=20
OPS_RECORDINGS_CRIT_GB=50
OPS_SIDEKIQ_STALE_SECS=300
OPS_LOG_SUBSCRIBER=false
SENTRY_DSN=
# Stream autoscale — token Hetzner da compilare; kill-switch cloud off finché WG non è ok
HCLOUD_TOKEN=
HCLOUD_LOCATION=nbg1
HCLOUD_SERVER_TYPE=cpx12
HCLOUD_IMAGE=debian-12
HCLOUD_SSH_KEY=matchlivetv-stream-hetzner
HCLOUD_NETWORK_ID=
HCLOUD_USER_DATA_FILE=/opt/matchlivetv/infra/stream-node/cloud-init.yaml
STREAM_DNS_ZONE=mltv-stream.net
STREAM_DNS_TTL=60
STREAM_CLOUD_DNS_SUFFIX=mltv-stream.net
STREAM_CLOUD_MAX_PUBLISHERS=4
STREAM_NODE_ENV=collaudo
STREAM_CLOUD_PROVIDER=local_lab
RELAY_MAX_CONCURRENT=4
YOUTUBE_RELAY_WORKER=1
STREAM_AUTOSCALE_ENABLED=0
STREAM_AUTOSCALE_KIND=lab
STREAM_AUTOSCALE_ALLOW_CLOUD=0
STREAM_AUTOSCALE_SOFT_FREE_SLOTS=2
STREAM_AUTOSCALE_WARM_SPARE=1
STREAM_AUTOSCALE_IDLE_MINUTES=30
STREAM_AUTOSCALE_MAX_NODES=5
STREAM_AUTOSCALE_NODE_EUR_PER_HOUR=0.015
STREAM_AUTOSCALE_MONTHLY_BUDGET_EUR=40
STREAM_OVERFLOW_ORPHAN_HOURS=3
+32
View File
@@ -104,3 +104,35 @@ OPS_LOG_SUBSCRIBER=false
# Sentry (opzionale — error tracking produzione)
SENTRY_DSN=
SENTRY_TRACES_SAMPLE_RATE=0.1
# --- Stream autoscale (Hetzner Cloud + DNS mltv-stream.net) ---
# Token progetto matchlivetv-stream (Read & Write). Non committare il valore reale.
HCLOUD_TOKEN=
HCLOUD_LOCATION=nbg1
HCLOUD_SERVER_TYPE=cpx12
HCLOUD_IMAGE=debian-12
HCLOUD_SSH_KEY=matchlivetv-stream-hetzner
HCLOUD_NETWORK_ID=
HCLOUD_USER_DATA_FILE=/opt/matchlivetv/infra/stream-node/cloud-init.yaml
STREAM_DNS_ZONE=mltv-stream.net
STREAM_DNS_TTL=60
STREAM_CLOUD_DNS_SUFFIX=mltv-stream.net
STREAM_CLOUD_MAX_PUBLISHERS=4
STREAM_NODE_ENV=prod
# Lab locale di default; il bottone admin "Provisiona nodo Hetzner" usa comunque i provider hetzner.
STREAM_CLOUD_PROVIDER=local_lab
RELAY_MAX_CONCURRENT=4
YOUTUBE_RELAY_WORKER=1
# Autoscaler (kill-switch: 0 finché WireGuard/smoke Cloud non sono OK — NON abilitare in prod senza lab)
STREAM_AUTOSCALE_ENABLED=0
STREAM_AUTOSCALE_KIND=lab
STREAM_AUTOSCALE_ALLOW_CLOUD=0
STREAM_AUTOSCALE_SOFT_FREE_SLOTS=2
STREAM_AUTOSCALE_WARM_SPARE=1
STREAM_AUTOSCALE_IDLE_MINUTES=30
STREAM_AUTOSCALE_MAX_NODES=5
STREAM_AUTOSCALE_INTERVAL_SECS=60
STREAM_AUTOSCALE_NODE_EUR_PER_HOUR=0.015
STREAM_AUTOSCALE_MONTHLY_BUDGET_EUR=40
STREAM_OVERFLOW_ORPHAN_HOURS=3
+35
View File
@@ -0,0 +1,35 @@
# Collaudo Proxmox (es. 192.168.1.157) — overlay su docker-compose.prod.yml
#
# Uso:
# export COMPOSE_FILE=docker-compose.prod.yml:docker-compose.collaudo.yml
# docker compose --env-file .env up -d --build
#
# Porte host MediaMTX diverse da produzione per non competere sullo stesso IP pubblico:
# prod WAN :1935 → 192.168.1.146
# collaudo LAN/WAN opzionale :11935 → 192.168.1.157
# HTTP resta :3000 (NPM → collaudo.matchlivetv.it); IP LAN diversi = nessun conflitto.
services:
mediamtx:
ports: !override
- "11935:1935" # RTMP collaudo (NON usare WAN :1935 di produzione)
- "18888:8888" # HLS diretto opzionale; in genere basta /hls via edge+NPM
# Passa tutto .env (HCLOUD_*, STREAM_*, …) ai processi Rails/Sidekiq
# Monta cloud-init: HCLOUD_USER_DATA_FILE punta a path host non presente nell'immagine.
rails:
env_file:
- .env
volumes:
- ./stream-node:/opt/matchlivetv/infra/stream-node:ro
sidekiq:
env_file:
- .env
volumes:
- ./stream-node:/opt/matchlivetv/infra/stream-node:ro
# ntfy collaudo su porta diversa (evita confusione se si punta per sbaglio l'host)
ntfy:
ports: !override
- "18090:80"
+2
View File
@@ -135,6 +135,7 @@ services:
- ${MATCHLIVETV_VIDEOS_ROOT:-/media/videos/matchlivetv}/recordings:/recordings
- ${MATCHLIVETV_VIDEOS_ROOT:-/media/videos/matchlivetv}/active_storage:/app/storage
- ${MATCHLIVETV_VIDEOS_ROOT:-/media/videos/matchlivetv}/log:/app/log
- ${HCLOUD_USER_DATA_HOST_DIR:-./stream-node}:/opt/matchlivetv/infra/stream-node:ro
healthcheck:
test: ["CMD", "curl", "-f", "http://127.0.0.1:3000/up"]
interval: 15s
@@ -230,6 +231,7 @@ services:
- ${MATCHLIVETV_VIDEOS_ROOT:-/media/videos/matchlivetv}/recordings:/recordings
- ${MATCHLIVETV_VIDEOS_ROOT:-/media/videos/matchlivetv}/active_storage:/app/storage
- ${MATCHLIVETV_VIDEOS_ROOT:-/media/videos/matchlivetv}/log:/app/log
- ${HCLOUD_USER_DATA_HOST_DIR:-./stream-node}:/opt/matchlivetv/infra/stream-node:ro
garage:
image: dxflrs/garage:v1.0.1
+5 -5
View File
@@ -5,14 +5,14 @@ set -euo pipefail
ROOT="$(cd "$(dirname "$0")/.." && pwd)"
cd "$ROOT"
export COMPOSE_FILE="docker-compose.prod.yml"
export COMPOSE_FILE="${COMPOSE_FILE:-docker-compose.prod.yml}"
export ENV_FILE="${ROOT}/.env"
export CREDS_FILE="${ROOT}/garage/prod-credentials.env"
bash "${ROOT}/scripts/ensure_garage_prod_config.sh"
echo "Avvio Garage..."
docker compose -f docker-compose.prod.yml --env-file .env up -d garage
echo "Avvio Garage (COMPOSE_FILE=${COMPOSE_FILE})..."
docker compose --env-file .env up -d garage
videos_root="${MATCHLIVETV_VIDEOS_ROOT:-/media/videos/matchlivetv}"
capacity_df="${videos_root}"
@@ -27,8 +27,8 @@ echo "Capacità nodo Garage: ${GARAGE_NODE_CAPACITY} (disco ~${avail_gb}G, riser
bash "${ROOT}/scripts/setup_garage_replays.sh"
echo "Riavvio Rails e Sidekiq..."
docker compose -f docker-compose.prod.yml --env-file .env up -d rails sidekiq
docker compose --env-file .env up -d rails sidekiq
echo "Verifica storage:"
docker compose -f docker-compose.prod.yml --env-file .env exec -T rails bundle exec rails runner \
docker compose --env-file .env exec -T rails bundle exec rails runner \
'puts({local: MatchLiveTv.replay_storage_local?, backend: Recordings::Storage.new.backend_name}.inspect)'
+78
View File
@@ -0,0 +1,78 @@
#cloud-config
# Nodo stream Hetzner: MediaMTX + ffmpeg.
# HCLOUD_USER_DATA_FILE=.../infra/stream-node/cloud-init.yaml
package_update: true
packages:
- docker.io
- ffmpeg
- curl
- apparmor
write_files:
- path: /opt/stream-node/mediamtx.yml
permissions: "0644"
content: |
logLevel: info
# Allinea a infra/mediamtx.yml: API raggiungibile anche da IP non-localhost
# (in prod restringere via WireGuard / firewall Hetzner).
authInternalUsers:
- user: any
pass:
ips: []
permissions:
- action: api
- action: metrics
- action: publish
- action: read
- action: playback
api: yes
apiAddress: :9997
rtmp: yes
rtmpAddress: :1935
hls: yes
hlsAddress: :8888
paths:
all_others:
- path: /opt/stream-node/bootstrap.sh
permissions: "0755"
content: |
#!/bin/bash
set -euo pipefail
mkdir -p /recordings /slates
systemctl enable --now docker
# aspetta il socket docker
for i in $(seq 1 60); do
if docker info >/dev/null 2>&1; then break; fi
sleep 2
done
docker pull bluenviron/mediamtx:latest
docker rm -f mediamtx 2>/dev/null || true
# Slate offline (alwaysAvailable) — richiesta da Mediamtx::Client#create_path
# MediaMTX 1.20+: il publisher RTMP deve matchare canali audio della slate (stereo).
mkdir -p /slates
if [ ! -f /slates/offline.mp4 ]; then
ffmpeg -y -f lavfi -i color=c=black:s=1280x720:d=2 -f lavfi -i anullsrc=r=44100:cl=stereo \
-c:v libx264 -t 2 -pix_fmt yuv420p -c:a aac -ac 2 -shortest /slates/offline.mp4
fi
# Debian cloud image: docker.io senza apparmor_parser → serve unconfined o pkg apparmor
docker run -d --name mediamtx --restart unless-stopped --network host \
--security-opt apparmor=unconfined \
-v /opt/stream-node/mediamtx.yml:/mediamtx.yml:ro \
-v /recordings:/recordings \
-v /slates:/slates:ro \
bluenviron/mediamtx:latest /mediamtx.yml
# smoke locale
for i in $(seq 1 30); do
if curl -fsS http://127.0.0.1:9997/v3/paths/list >/dev/null; then
echo "mediamtx ready" | tee /opt/stream-node/ready
exit 0
fi
sleep 2
done
echo "mediamtx NOT ready" | tee /opt/stream-node/ready
docker logs mediamtx || true
exit 1
runcmd:
- [bash, /opt/stream-node/bootstrap.sh]
@@ -17,11 +17,16 @@ import org.junit.runner.RunWith
/**
* E2E sul simulatore: login nuova partita wizard 3 step schermata diretta.
* Usa le stringhe dell'app (locale del device), non testo hardcodato IT/EN.
*/
@RunWith(AndroidJUnit4::class)
class E2EWizardFlowTest {
private lateinit var device: UiDevice
private val pkg = "com.matchlivetv.match_live_tv"
private val ctx by lazy { InstrumentationRegistry.getInstrumentation().targetContext }
private fun s(id: Int): String = ctx.getString(id)
private fun su(id: Int): String = s(id).uppercase()
@Before
fun setUp() {
@@ -32,30 +37,42 @@ class E2EWizardFlowTest {
@Test
fun login_newMatch_wizard_reachesBroadcastScreen() {
waitForAnyText("Email", "ACCEDI", timeoutMs = 45_000)
waitForAnyText(s(R.string.login_email), su(R.string.login_submit), timeoutMs = 45_000)
fillLogin()
waitForText("NUOVA PARTITA", timeoutMs = 45_000)
tapClickableText("NUOVA PARTITA")
waitForText("Avvia subito", timeoutMs = 15_000)
tapClickableText("Avvia subito")
waitForText("01 · Partita", timeoutMs = 45_000)
waitForText(su(R.string.matches_new), timeoutMs = 45_000)
tapClickableText(su(R.string.matches_new))
waitForText(s(R.string.sheet_quick_option), timeoutMs = 15_000)
tapClickableText(s(R.string.sheet_quick_option))
waitForText(s(R.string.wizard_step_title_match), timeoutMs = 45_000)
scrollDown()
tapClickableText("AVANTI >")
waitForText("02 · Trasmissione", timeoutMs = 45_000)
waitForText("Piattaforma", timeoutMs = 30_000)
tapClickableText(s(R.string.wizard_action_next))
waitForText(s(R.string.wizard_step_title_transmission), timeoutMs = 45_000)
waitForText(s(R.string.wizard_transmission_platform_title), timeoutMs = 30_000)
scrollDown()
tapClickableText("AVANTI >")
waitForText("03 · Test rete", timeoutMs = 45_000)
waitForText("AVVIA TEST RETE", timeoutMs = 30_000)
tapClickableText("AVVIA TEST RETE")
waitForText("INIZIA >", timeoutMs = 30_000)
waitUntilEnabled("INIZIA >", timeoutMs = 25_000)
tapClickableText(s(R.string.wizard_action_next))
waitForText(s(R.string.wizard_step_title_network), timeoutMs = 45_000)
waitForText(s(R.string.wizard_network_test_start_label), timeoutMs = 30_000)
tapClickableText(s(R.string.wizard_network_test_start_label))
waitForText(s(R.string.wizard_action_start), timeoutMs = 30_000)
waitUntilEnabled(s(R.string.wizard_action_start), timeoutMs = 25_000)
scrollDown()
tapClickableText("INIZIA >")
val diretta = waitForText("Diretta", timeoutMs = 60_000)
assertNotNull(diretta)
assertNotNull(waitForText("TERMINA DIRETTA", timeoutMs = 30_000))
assertNotNull(waitForText("CHIUDI SET", timeoutMs = 20_000))
tapClickableText(s(R.string.wizard_action_start))
waitForAnyText(
s(R.string.broadcast_status_live),
s(R.string.broadcast_status_connecting),
s(R.string.broadcast_status_reconnecting),
timeoutMs = 60_000,
)
// CLOSE SET è un SideIconButton: in hierarchy compare come content-desc, non sempre come text.
assertTrue(
"Schermata diretta non pronta (né CLOSE SET né End live)",
waitForTextOrDesc(
timeoutMs = 30_000,
s(R.string.broadcast_close_set_button),
s(R.string.score_action_close_set),
s(R.string.broadcast_terminate_cd),
),
)
}
private fun grantRuntimePermissions() {
@@ -69,11 +86,10 @@ class E2EWizardFlowTest {
}
private fun launchApp() {
val context = InstrumentationRegistry.getInstrumentation().targetContext
val intent = context.packageManager.getLaunchIntentForPackage(pkg)?.apply {
val intent = ctx.packageManager.getLaunchIntentForPackage(pkg)?.apply {
addFlags(Intent.FLAG_ACTIVITY_CLEAR_TASK or Intent.FLAG_ACTIVITY_NEW_TASK)
} ?: error("Launch intent mancante per $pkg")
context.startActivity(intent)
ctx.startActivity(intent)
device.wait(Until.hasObject(By.pkg(pkg).depth(0)), 15_000)
}
@@ -81,8 +97,10 @@ class E2EWizardFlowTest {
val fields = device.wait(Until.findObjects(By.clazz("android.widget.EditText")), 15_000)
if (fields.size < 2) error("Campi login non trovati (${fields.size})")
pasteIntoField(fields[0], "coach@matchlivetv.test")
pasteIntoField(fields[1], "password123")
pasteIntoField(fields[1], "Password123")
device.pressKeyCode(KeyEvent.KEYCODE_ENTER)
// Preferisci il bottone (ultimo match uppercase), non il titolo.
tapLastMatchingText(su(R.string.login_submit))
device.waitForIdle()
}
@@ -111,6 +129,34 @@ class E2EWizardFlowTest {
error("Nessuno dei testi trovato: ${texts.joinToString()}")
}
private fun waitForTextOrDesc(timeoutMs: Long, vararg labels: String): Boolean {
val deadline = SystemClock.elapsedRealtime() + timeoutMs
while (SystemClock.elapsedRealtime() < deadline) {
for (label in labels) {
if (device.hasObject(By.text(label)) || device.hasObject(By.desc(label))) return true
}
SystemClock.sleep(250)
}
return false
}
private fun tapLastMatchingText(text: String) {
val nodes = device.findObjects(By.text(text))
val target = nodes.lastOrNull { it.isClickable }
?: nodes.lastOrNull()?.let { label ->
var node: UiObject2? = label
repeat(6) {
val current = node ?: return@repeat
if (current.isClickable) return@let current
node = current.parent
}
label
}
?: error("Testo non trovato: $text")
target.click()
device.waitForIdle()
}
private fun tapClickableText(text: String) {
device.findObject(By.text(text).clickable(true))?.let { node ->
node.click()
+11
View File
@@ -0,0 +1,11 @@
#!/usr/bin/env bash
# Helper collaudo Proxmox: forza sempre prod + overlay porte isolate.
set -euo pipefail
ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
INFRA="${MATCHLIVETV_INFRA:-$ROOT/infra}"
if [ -d /opt/matchlivetv/infra ]; then
INFRA=/opt/matchlivetv/infra
fi
cd "$INFRA"
export COMPOSE_FILE="${COMPOSE_FILE:-docker-compose.prod.yml:docker-compose.collaudo.yml}"
exec docker compose --env-file .env "$@"