Loading...
Searching...
No Matches
GpuUtils.hpp
1#pragma once
2
3#include <avnd/introspection/gfx.hpp>
4#if SCORE_PLUGIN_GFX
5#include <Process/ExecutionContext.hpp>
6
7#include <Crousti/File.hpp>
8#include <Crousti/GppCoroutines.hpp>
9#include <Crousti/GppShaders.hpp>
10#include <Crousti/MessageBus.hpp>
11#include <Crousti/SceneConcepts.hpp>
12#include <Crousti/TextureConversion.hpp>
13#include <Crousti/TextureFormat.hpp>
14#include <Crousti/WorkerBinding.hpp>
15#include <Gfx/GfxExecNode.hpp>
16#include <Gfx/Graph/Node.hpp>
17#include <Gfx/Graph/OutputNode.hpp>
18#include <Gfx/Graph/RenderList.hpp>
19#include <Gfx/Graph/RenderState.hpp>
20
21#include <score/tools/ThreadPool.hpp>
22
23#include <ossia/detail/hash_map.hpp>
24#include <ossia/detail/lockfree_queue.hpp>
25#include <ossia/detail/small_flat_map.hpp>
26#include <ossia/detail/small_vector.hpp>
27
28#include <ossia-qt/invoke.hpp>
29
30#include <QCoreApplication>
31#include <QTimer>
32#include <QtGui/private/qrhi_p.h>
33
34#include <avnd/binding/ossia/metadatas.hpp>
35#include <avnd/binding/ossia/port_run_postprocess.hpp>
36#include <avnd/binding/ossia/port_run_preprocess.hpp>
37#include <avnd/binding/ossia/soundfiles.hpp>
38#include <avnd/concepts/parameter.hpp>
39#include <avnd/introspection/input.hpp>
40#include <avnd/introspection/output.hpp>
41#include <fmt/format.h>
42#include <gpp/layout.hpp>
43#include <halp/texture.hpp>
44
45#include <score_plugin_avnd_export.h>
46
47#include <algorithm>
48#include <atomic>
49
50namespace oscr
51{
63struct GpuWorker
64{
65 struct WorkerResults
66 {
67 struct Result
68 {
69 uint64_t order{};
70 std::function<void()> apply;
71 };
72 ossia::mpmc_queue<Result> queue;
73 std::atomic<uint64_t> pushed{};
74 std::atomic<int> inflight{};
75
76 void push(std::function<void()> f)
77 {
78 queue.enqueue(Result{pushed.fetch_add(1, std::memory_order_relaxed), std::move(f)});
79 }
80
81 void drain()
82 {
83 ossia::small_vector<Result, 4> batch;
84 Result r;
85 while(queue.try_dequeue(r))
86 batch.push_back(std::move(r));
87 std::ranges::sort(batch, {}, &Result::order);
88 for(auto& res : batch)
89 res.apply();
90 }
91
92 // Waits for the jobs in flight, and for those that the results applied
93 // meanwhile requested.
94 void settle()
95 {
96 for(;;)
97 {
98 for(int n = inflight.load(); n > 0; n = inflight.load())
99 inflight.wait(n);
100 if(queue.size_approx() == 0)
101 return;
102 drain();
103 }
104 }
105 };
106
109 struct Delivery
110 {
111 std::weak_ptr<WorkerResults> results;
112
113 struct Job
114 {
115 std::weak_ptr<WorkerResults> results;
116 explicit Job(std::weak_ptr<WorkerResults> r) noexcept
117 : results{std::move(r)}
118 {
119 if(auto p = results.lock())
120 p->inflight.fetch_add(1);
121 }
122 Job(Job&& other) noexcept = default;
123 Job& operator=(Job&&) = delete;
124 ~Job()
125 {
126 if(auto p = results.lock())
127 {
128 p->inflight.fetch_sub(1);
129 p->inflight.notify_all();
130 }
131 }
132
133 template <typename F>
134 void operator()(F&& apply)
135 {
136 auto r = results.lock();
137 if(!r)
138 return;
139 r->push(std::forward<F>(apply));
140 ossia::qt::run_async(QCoreApplication::instance(), [wr = results] {
141 if(auto r = wr.lock())
142 r->drain();
143 });
144 }
145 };
146 Job begin() const noexcept { return Job{results}; }
147 };
148
151 void drainWorker(bool settle = false) const
152 {
153 if(!m_workerResults)
154 return;
155 if(settle)
156 m_workerResults->settle();
157 else
158 m_workerResults->drain();
159 }
160
161 template <typename T>
162 void initWorker(this auto& self, std::shared_ptr<T>& state) noexcept
163 {
164 if constexpr(avnd::has_worker<T>)
165 {
166 if(!self.m_workerResults)
167 self.m_workerResults = std::make_shared<WorkerResults>();
168
169 bind_worker(*state, std::weak_ptr{state}, Delivery{self.m_workerResults});
170 }
171 }
172
173 mutable std::shared_ptr<WorkerResults> m_workerResults;
174};
175
176#if defined(OSCR_HAS_MMAP_FILE_STORAGE)
177// The file ports keep string_views into the raw_file_data, and there is one
178// object instance per renderer (one renderer per RenderList, that is per
179// output) and per instance in CustomGpuRenderer. Storing the handles on the
180// node would thus free, on every load, the memory that the ports of all the
181// other instances still point to.
182template <typename T>
183struct GpuRendererFiles
184{
185 template <std::size_t N, std::size_t NField>
186 void file_loaded(
187 T& state, const std::shared_ptr<oscr::raw_file_data>& hdl,
188 avnd::predicate_index<N> pred, avnd::field_index<NField> field)
189 {
190 m_rawfiles[&state].load(state, hdl, pred, field);
191 }
192
193 void releaseFiles() noexcept { m_rawfiles.clear(); }
194
195private:
196 ossia::hash_map<const T*, oscr::raw_file_storage<T>> m_rawfiles;
197};
198
199template <typename T>
200 requires(avnd::raw_file_input_introspection<T>::size == 0)
201struct GpuRendererFiles<T>
202{
203 void releaseFiles() noexcept { }
204};
205#else
206template <typename T>
207struct GpuRendererFiles
208{
209 void releaseFiles() noexcept { }
210};
211#endif
212
220struct SCORE_PLUGIN_AVND_EXPORT GpuMessageState
221{
222 score::gfx::Message message;
223 std::vector<uint32_t> generation;
224
225 void process(score::gfx::Message&& msg) noexcept;
226};
227
228template <typename GpuNodeRenderer, typename Node>
229struct GpuProcessIns
230{
231 GpuProcessIns(
232 GpuNodeRenderer& gpu, Node& state, const GpuMessageState& prev,
233 const GpuMessageState& cur, const score::DocumentContext& ctx) noexcept
234 : gpu{gpu}
235 , state{state}
236 , prev_mess{prev.message}
237 , mess{cur.message}
238 , prev_gen{prev.generation}
239 , gen{cur.generation}
240 , ctx{ctx}
241 {
242 }
243
244 GpuNodeRenderer& gpu;
245 Node& state;
246 const score::gfx::Message& prev_mess;
247 const score::gfx::Message& mess;
248 const std::vector<uint32_t>& prev_gen;
249 const std::vector<uint32_t>& gen;
250 const score::DocumentContext& ctx;
251
252 bool can_process_message(std::size_t N)
253 {
254 if(mess.input.size() <= N)
255 return false;
256
257 if(prev_mess.input.size() == mess.input.size())
258 {
259 auto& prev = prev_mess.input[N];
260 auto& next = mess.input[N];
261 if(prev.index() == 1 && next.index() == 1)
262 {
263 if(ossia::get<ossia::value>(prev) == ossia::get<ossia::value>(next))
264 {
265 return false;
266 }
267 }
268 }
269 return true;
270 }
271
272 // The node's message keeps the last value of every input, so the one-shot values in it are
273 // applied once per arrival rather than once per value: a bang equal to the previous one is a
274 // new event, the same one still sitting in the message is not.
275 bool received_event(std::size_t N) const noexcept
276 {
277 if(mess.input.size() <= N || gen.size() <= N)
278 return false;
279 return prev_gen.size() <= N || prev_gen[N] != gen[N];
280 }
281
282 template <avnd::parameter_port Field, std::size_t NField>
283 void operator()(Field& t, avnd::field_index<NField> field_index)
284 {
285 if constexpr(avnd::optional_ish<decltype(Field::value)>)
286 {
287 // Mirrors the CPU path, which sees each port's data stream once per tick.
288 if(!received_event(NField))
289 return;
290 }
291 else if(!can_process_message(field_index))
292 {
293 return;
294 }
295
296 if(auto val = ossia::get_if<ossia::value>(&mess.input[field_index]))
297 {
298 oscr::from_ossia_value(t, *val, t.value);
299 if_possible(t.update(state));
300 }
301 }
302
303#if OSCR_HAS_MMAP_FILE_STORAGE
304 template <avnd::raw_file_port Field, std::size_t NField>
305 void operator()(Field& t, avnd::field_index<NField> field_index)
306 {
307 // FIXME we should be loading a file there
308 using file_ports = avnd::raw_file_input_introspection<Node>;
309
310 if(!can_process_message(field_index))
311 return;
312
313 auto val = ossia::get_if<ossia::value>(&mess.input[field_index]);
314 if(!val)
315 return;
316
317 static constexpr bool has_text = requires { decltype(Field::file)::text; };
318 static constexpr bool has_mmap = requires { decltype(Field::file)::mmap; };
319
320 // First we can load it directly since execution hasn't started yet
321 if(auto hdl = loadRawfile(*val, ctx, has_text, has_mmap))
322 {
323 static constexpr auto N = file_ports::field_index_to_index(NField);
324 if constexpr(avnd::port_can_process<Field>)
325 {
326 // FIXME also do it when we get a run-time message from the exec engine,
327 // OSC, etc
328 auto func = executePortPreprocess<Field>(*hdl);
329 gpu.file_loaded(
330 state, hdl, avnd::predicate_index<N>{}, avnd::field_index<NField>{});
331 if(func)
332 func(state);
333 }
334 else
335 {
336 gpu.file_loaded(
337 state, hdl, avnd::predicate_index<N>{}, avnd::field_index<NField>{});
338 }
339 }
340 }
341#endif
342
343 template <avnd::buffer_port Field, std::size_t NField>
344 void operator()(Field& t, avnd::field_index<NField> field_index)
345 {
346 if(!can_process_message(field_index))
347 return;
348
349 using node_type = std::remove_cvref_t<decltype(gpu.node())>;
350 auto& node = const_cast<node_type&>(gpu.node());
351 if(field_index >= mess.input.size())
352 return;
353 auto val = ossia::get_if<ossia::render_target_spec>(&mess.input[field_index]);
354 if(!val)
355 return;
356 static_cast<score::gfx::Node&>(node).process(int32_t(NField), *val);
357 }
358
359 template <avnd::texture_port Field, std::size_t NField>
360 void operator()(Field& t, avnd::field_index<NField> field_index)
361 {
362 if(!can_process_message(field_index))
363 return;
364
365 using node_type = std::remove_cvref_t<decltype(gpu.node())>;
366 auto& node = const_cast<node_type&>(gpu.node());
367 if(field_index >= mess.input.size())
368 return;
369 auto val = ossia::get_if<ossia::render_target_spec>(&mess.input[field_index]);
370 if(!val)
371 return;
372 static_cast<score::gfx::Node&>(node).process(int32_t(NField), *val);
373 }
374
375 template <avnd::geometry_port Field, std::size_t NField>
376 void operator()(Field& t, avnd::field_index<NField> field_index)
377 {
378 // Intentional no-op: geometry is not in the control message, it flows
379 // through geometry_inputs_storage::readInputGeometries. The empty body
380 // keeps GpuProcessIns instantiable for nodes whose input list contains
381 // geometry fields, which would otherwise hit the `= delete` catch-all
382 // below.
383 }
384
385 template <scene_port Field, std::size_t NField>
386 void operator()(Field& t, avnd::field_index<NField> field_index)
387 {
388 // Intentional no-op — same reasoning as the geometry_port overload above.
389 // Scene data flows through scene_inputs_storage / scene_outputs_storage.
390 }
391
392 void operator()(auto& t, auto field_index) = delete;
393};
394
395struct GpuControlIns
396{
397 template <typename Self, typename Node_T>
398 static void processControlIn(
399 Self& self, Node_T& state, GpuMessageState& renderer_mess,
400 const GpuMessageState& mess, const score::DocumentContext& ctx) noexcept
401 {
402 // Apply the controls
403 avnd::input_introspection<Node_T>::for_all_n(
404 avnd::get_inputs<Node_T>(state),
405 GpuProcessIns<Self, Node_T>{self, state, renderer_mess, mess, ctx});
406 renderer_mess = mess;
407 }
408
409 // Once the processor has run, the impulses and other one-shot values it received this tick are
410 // consumed, exactly as oscr::process_after_run does on the CPU path. Without this an impulse
411 // would stay set until the next message touches the port.
412 template <typename Node_T>
413 static void clearControlIn(Node_T& state) noexcept
414 {
415 avnd::parameter_input_introspection<Node_T>::for_all(
416 avnd::get_inputs<Node_T>(state), []<typename Field>(Field& t) {
417 if constexpr(avnd::optional_ish<decltype(Field::value)>)
418 t.value.reset();
419 });
420 }
421};
422
423struct GpuControlOuts
424{
425 std::weak_ptr<Execution::ExecutionCommandQueue> queue;
426 Gfx::exec_controls control_outs;
427
428 int64_t instance{};
429
430 template <typename Node_T>
431 void processControlOut(Node_T& state) const noexcept
432 {
433 if(!this->control_outs.empty())
434 {
435 auto q = this->queue.lock();
436 if(!q)
437 return;
438 auto& qq = *q;
439 int parm_k = 0;
440 avnd::parameter_output_introspection<Node_T>::for_all(
441 avnd::get_outputs(state), [&]<avnd::parameter_port T>(const T& t) {
442 qq.enqueue([v = oscr::to_ossia_value(t, t.value),
443 port = control_outs[parm_k]]() mutable {
444 std::swap(port->value, v);
445 port->changed = true;
446 });
447
448 parm_k++;
449 });
450 }
451 }
452};
453
454template <typename T>
455struct SCORE_PLUGIN_AVND_EXPORT GpuNodeElements
456{
457 [[no_unique_address]] oscr::soundfile_storage<T> soundfiles;
458
459 [[no_unique_address]] oscr::midifile_storage<T> midifiles;
460
461 // Raw file handles are stored per-state in the renderers, see GpuRendererFiles
462};
463
464struct SCORE_PLUGIN_AVND_EXPORT CustomGfxNodeBase : score::gfx::NodeModel
465{
466 explicit CustomGfxNodeBase(const score::DocumentContext& ctx)
467 : score::gfx::NodeModel{}
468 , m_ctx{ctx}
469 {
470 }
471 virtual ~CustomGfxNodeBase();
472 const score::DocumentContext& m_ctx;
473 GpuMessageState last_message;
474 void process(score::gfx::Message&& msg) override;
476};
477struct SCORE_PLUGIN_AVND_EXPORT CustomGfxOutputNodeBase : score::gfx::OutputNode
478{
479 virtual ~CustomGfxOutputNodeBase();
480
481 GpuMessageState last_message;
482 void process(score::gfx::Message&& msg) override;
483};
484struct SCORE_PLUGIN_AVND_EXPORT CustomGpuNodeBase
486 , GpuWorker
487 , GpuControlIns
488 , GpuControlOuts
489{
490 CustomGpuNodeBase(
491 std::weak_ptr<Execution::ExecutionCommandQueue>&& q, Gfx::exec_controls&& ctls,
492 const score::DocumentContext& ctx)
493 : GpuControlOuts{std::move(q), std::move(ctls)}
494 , m_ctx{ctx}
495 {
496 }
497
498 virtual ~CustomGpuNodeBase() = default;
499
500 const score::DocumentContext& m_ctx;
501 QString vertex, fragment, compute;
502 GpuMessageState last_message;
503 void process(score::gfx::Message&& msg) override;
504};
505
506struct SCORE_PLUGIN_AVND_EXPORT CustomGpuOutputNodeBase
508 , GpuWorker
509 , GpuControlIns
510 , GpuControlOuts
511{
515 static constexpr QSize defaultRenderSize{200, 200};
516
517 CustomGpuOutputNodeBase(
518 std::weak_ptr<Execution::ExecutionCommandQueue> q, Gfx::exec_controls&& ctls,
519 const score::DocumentContext& ctx);
520 virtual ~CustomGpuOutputNodeBase();
521
522 const score::DocumentContext& m_ctx;
523 std::weak_ptr<score::gfx::RenderList> m_renderer{};
524 std::shared_ptr<score::gfx::RenderState> m_renderState{};
525
526 QString vertex, fragment, compute;
527 GpuMessageState last_message;
528 void process(score::gfx::Message&& msg) override;
530
531 void setRenderer(std::shared_ptr<score::gfx::RenderList>) override;
532 score::gfx::RenderList* renderer() const override;
533
534 void startRendering() override;
535 void render() override;
536 void stopRendering() override;
537 bool canRender() const override;
538 void onRendererChange() override;
539
540 void createOutput(score::gfx::OutputConfiguration) override;
541
542 void destroyOutput() override;
543 std::shared_ptr<score::gfx::RenderState> renderState() const override;
544
545 Configuration configuration() const noexcept override;
546};
547
548template <typename Node_T, typename Node>
549void prepareNewState(std::shared_ptr<Node_T>& eff, const Node& parent)
550{
551 if constexpr(avnd::has_worker<Node_T>)
552 {
553 parent.initWorker(eff);
554 }
555 if constexpr(avnd::has_processor_to_gui_bus<Node_T>)
556 {
557 auto& process = parent.processModel;
558 eff->send_message = [ptr = QPointer{&process}](auto&& b) mutable {
559 // FIXME right now all the rendering is done in the UI thread, which is very MEH
560 // this->in_edit([&process, bb = std::move(b)]() mutable {
561
562 if(ptr && ptr->to_ui)
563 MessageBusSender{ptr->to_ui}(std::move(b));
564 // });
565 };
566
567 // FIXME GUI -> engine. See executor.hpp
568 }
569
570 avnd::init_controls(*eff);
571
572 if constexpr(avnd::can_prepare<Node_T>)
573 {
574 if constexpr(avnd::function_reflection<&Node_T::prepare>::count == 1)
575 {
576 using prepare_type = avnd::first_argument<&Node_T::prepare>;
577 prepare_type t;
578 if_possible(t.instance = parent.instance);
579 eff->prepare(t);
580 }
581 else
582 {
583 eff->prepare();
584 }
585 }
586}
587
588struct port_to_type_enum
589{
590 template <std::size_t I, avnd::buffer_port F>
591 constexpr auto operator()(avnd::field_reflection<I, F> p)
592 {
593 return score::gfx::Types::Buffer;
594 }
595
596 template <std::size_t I, avnd::cpu_texture_port F>
597 constexpr auto operator()(avnd::field_reflection<I, F> p)
598 {
599 using texture_type = std::remove_cvref_t<decltype(F::texture)>;
600 return (avnd::cpu_fixed_format_texture<texture_type> || avnd::cpu_dynamic_format_texture<texture_type>)
601 ? score::gfx::Types::Image
602 : score::gfx::Types::Buffer;
603 }
604
605 template <std::size_t I, avnd::gpu_texture_port F>
606 constexpr auto operator()(avnd::field_reflection<I, F> p)
607 {
608 return score::gfx::Types::Image;
609 }
610
611 template <std::size_t I, avnd::sampler_port F>
612 constexpr auto operator()(avnd::field_reflection<I, F> p)
613 {
614 return score::gfx::Types::Image;
615 }
616 template <std::size_t I, avnd::image_port F>
617 constexpr auto operator()(avnd::field_reflection<I, F> p)
618 {
619 return score::gfx::Types::Image;
620 }
621 template <std::size_t I, avnd::attachment_port F>
622 constexpr auto operator()(avnd::field_reflection<I, F> p)
623 {
624 return score::gfx::Types::Image;
625 }
626 template <std::size_t I, avnd::gpu_render_target_output_port F>
627 constexpr auto operator()(avnd::field_reflection<I, F> p)
628 {
629 return score::gfx::Types::Image;
630 }
631
632 template <std::size_t I, avnd::geometry_port F>
633 constexpr auto operator()(avnd::field_reflection<I, F> p)
634 {
635 return score::gfx::Types::Geometry;
636 }
637 // Scene ports reuse Types::Geometry — a scene is a richer form of geometry.
638 template <std::size_t I, scene_port F>
639 requires(!avnd::geometry_port<F>)
640 constexpr auto operator()(avnd::field_reflection<I, F> p)
641 {
642 return score::gfx::Types::Geometry;
643 }
644 template <std::size_t I, avnd::mono_audio_port F>
645 constexpr auto operator()(avnd::field_reflection<I, F> p)
646 {
647 return score::gfx::Types::Audio;
648 }
649 template <std::size_t I, avnd::poly_audio_port F>
650 constexpr auto operator()(avnd::field_reflection<I, F> p)
651 {
652 return score::gfx::Types::Audio;
653 }
654 template <std::size_t I, avnd::int_parameter F>
655 constexpr auto operator()(avnd::field_reflection<I, F> p)
656 {
657 return score::gfx::Types::Int;
658 }
659 template <std::size_t I, avnd::enum_parameter F>
660 constexpr auto operator()(avnd::field_reflection<I, F> p)
661 {
662 return score::gfx::Types::Int;
663 }
664 template <std::size_t I, avnd::float_parameter F>
665 constexpr auto operator()(avnd::field_reflection<I, F> p)
666 {
667 return score::gfx::Types::Float;
668 }
669 template <std::size_t I, avnd::parameter_port F>
670 constexpr auto operator()(avnd::field_reflection<I, F> p)
671 {
672 using value_type = std::remove_cvref_t<decltype(F::value)>;
673
674 if constexpr(std::is_array_v<value_type>)
675 {
676 static constexpr int sz = sizeof(value_type) / sizeof(value_type{}[0]);
677 if constexpr(sz == 2)
678 {
679 return score::gfx::Types::Vec2;
680 }
681 else if constexpr(sz == 3)
682 {
683 return score::gfx::Types::Vec3;
684 }
685 else if constexpr(sz == 4)
686 {
687 return score::gfx::Types::Vec4;
688 }
689 }
690 else if constexpr(std::is_aggregate_v<value_type>)
691 {
692 static constexpr int sz = avnd::pfr::tuple_size_v<value_type>;
693 if constexpr(sz == 2)
694 {
695 return score::gfx::Types::Vec2;
696 }
697 else if constexpr(sz == 3)
698 {
699 return score::gfx::Types::Vec3;
700 }
701 else if constexpr(sz == 4)
702 {
703 return score::gfx::Types::Vec4;
704 }
705 }
706 return score::gfx::Types::Empty;
707 }
708 template <std::size_t I, typename F>
709 constexpr auto operator()(avnd::field_reflection<I, F> p)
710 {
711 return score::gfx::Types::Empty;
712 }
713};
714
715// Compile-time port flags derived from a field's declarative metadata.
716// Inspects:
717// - `texture_target` (texture_kind_of) — non-2D textures bypass the
718// local-RT allocation and grab the upstream texture directly.
719// - `samplable_depth` (samplable_depth_of) — opt-in to having the
720// framework allocate a sampleable depth attachment on the producing
721// edge's RT and expose its handle through `texture.depth_handle`,
722// mirroring the semantics CSF/ISF shaders get via "DEPTH": true.
723template <typename Field>
724constexpr score::gfx::Flag port_flags_for_field() noexcept
725{
726 if constexpr(avnd::gpu_texture_port<Field>)
727 {
728 constexpr auto kind = halp::texture_kind_of<Field>();
729 constexpr bool nonD2 = (kind != halp::texture_kind::texture_2d);
730 constexpr bool depth = halp::samplable_depth_of<Field>();
731 static_assert(
732 !(single_cable_port<Field> && (depth || nonD2)),
733 "single_cable is only supported on 2D texture inputs without depth");
734 if constexpr(single_cable_port<Field>)
735 return score::gfx::Flag::SingleCable;
736 else if constexpr(nonD2 && depth)
737 return score::gfx::Flag::GrabsFromSource | score::gfx::Flag::SamplableDepth;
738 else if constexpr(nonD2)
739 return score::gfx::Flag::GrabsFromSource;
740 else if constexpr(depth)
741 return score::gfx::Flag::SamplableDepth;
742 }
743 return score::gfx::Flag{};
744}
745
746// Map QRhi's depth-format taxonomy onto halp's depth_format_t.
747// The 4-arg subset matches every depth format score's createRenderTarget
748// can produce (today always D32F, but the API accepts the others).
749inline constexpr halp::gpu_texture::depth_format_t qrhiToHalpDepthFormat(
750 QRhiTexture::Format f) noexcept
751{
752 using D = halp::gpu_texture::depth_format_t;
753 switch(f)
754 {
755 case QRhiTexture::D16: return D::D16;
756 case QRhiTexture::D24: return D::D24;
757 case QRhiTexture::D24S8: return D::D24S8;
758 case QRhiTexture::D32F: return D::D32F;
759 default: break;
760 }
761 return D::D32F;
762}
763
764template <typename Node_T>
765inline void initGfxPorts(auto* self, auto& input, auto& output)
766{
767 avnd::input_introspection<Node_T>::for_all(
768 [self, &input]<typename Field, std::size_t I>(avnd::field_reflection<I, Field> f) {
769 static constexpr auto type = port_to_type_enum{}(f);
770 static constexpr auto flags = port_flags_for_field<Field>();
771 input.push_back(new score::gfx::Port{self, {}, type, flags, {}});
772 });
773 avnd::output_introspection<Node_T>::for_all(
774 [self,
775 &output]<typename Field, std::size_t I>(avnd::field_reflection<I, Field> f) {
776 static constexpr auto type = port_to_type_enum{}(f);
777 // port_flags_for_field encodes INPUT-side sink semantics
778 // (GrabsFromSource → "sample the upstream's texture directly";
779 // SamplableDepth → "ask the producer for a sampleable depth
780 // attachment"). Neither has any meaning on an OUTPUT port — emitting
781 // them here would make the graph treat this node's own output as if it
782 // grabbed from / sampled some upstream source. Outputs carry no such
783 // flags.
784 output.push_back(new score::gfx::Port{self, {}, type, score::gfx::Flag{}, {}});
785 });
786}
787
788static score::gfx::BufferView getInputBuffer(
789 score::gfx::RenderList& renderer, const score::gfx::Node& parent, int port_index)
790{
791 const auto& inputs = parent.input;
792 // SCORE_ASSERT(port_index == 0);
793 {
794 score::gfx::Port* p = inputs[port_index];
795 for(auto& edge : p->edges)
796 {
797 auto src_node = edge->source->node;
798 score::gfx::NodeRenderer* src_renderer = src_node->renderedNodes.at(&renderer);
799 if(src_renderer)
800 {
801 return src_renderer->bufferForOutput(*edge->source);
802 }
803 break;
804 }
805 }
806 return {};
807}
808
809
810static void readbackInputBuffer(
811 score::gfx::RenderList& renderer
812 , QRhiResourceUpdateBatch& res
813 , const score::gfx::Node& parent
814 , QRhiBufferReadbackResult& readback
815 , int port_index
816 )
817{
818 // FIXME: instead of doing this we could do the readback in the
819 // producer node and just read its bytearray once...
820 if(auto buf = getInputBuffer(renderer, parent, port_index))
821 {
822 readback = {};
823 res.readBackBuffer(buf.handle, buf.byte_offset, buf.byte_size, &readback);
824 }
825}
826
827static void recreateOutputBuffer(
828 score::gfx::RenderList& renderer, avnd::cpu_buffer auto& cpu_buf,
829 QRhiResourceUpdateBatch& res, score::gfx::BufferView& buf)
830{
831 const auto bytesize = avnd::get_bytesize(cpu_buf);
832 if(!buf.handle)
833 {
834 if(bytesize > 0)
835 {
836 buf.handle = renderer.state.rhi->newBuffer(
837 QRhiBuffer::Static,
839 *renderer.state.rhi,
840 QRhiBuffer::StorageBuffer | QRhiBuffer::VertexBuffer),
841 bytesize);
842 buf.handle->setName("GpuUtils::recreateOutputBuffer");
843 buf.byte_offset = 0;
844 buf.byte_size = bytesize;
845
846 buf.handle->create();
847 }
848 else
849 {
850 cpu_buf.changed = false;
851 return;
852 }
853 }
854 else if(buf.handle->size() != bytesize)
855 {
856 buf.handle->destroy();
857 buf.handle->setSize(bytesize);
858 buf.handle->create();
859 buf.byte_size = bytesize;
860 }
861}
862
863static void uploadOutputBuffer(
864 score::gfx::RenderList& renderer, avnd::cpu_buffer auto& cpu_buf,
865 QRhiResourceUpdateBatch& res, score::gfx::BufferView& rhi_buf)
866{
867 if(cpu_buf.changed)
868 {
869 recreateOutputBuffer(renderer, cpu_buf, res, rhi_buf);
871 &res, rhi_buf.handle, 0, cpu_buf.byte_size,
872 (const char*)avnd::get_bytes(cpu_buf));
873 cpu_buf.changed = false;
874 }
875}
876
877static void uploadOutputBuffer(
878 score::gfx::RenderList& renderer, avnd::gpu_buffer auto& gpu_buf,
879 QRhiResourceUpdateBatch& res, score::gfx::BufferView& rhi_buf)
880{
881 rhi_buf.handle = reinterpret_cast<QRhiBuffer*>(gpu_buf.handle);
882 rhi_buf.byte_size = gpu_buf.byte_size;
883 rhi_buf.byte_offset = gpu_buf.byte_offset;
884}
885
886template <typename T>
887struct geometry_inputs_storage;
888
889struct mesh_input_storage
890{
891 std::vector<QRhiBufferReadbackResult> readbacks;
892 std::vector<QRhiBuffer*> buffers;
893};
894struct geometry_input_storage
895{
896 ossia::geometry_spec spec;
897 std::vector<mesh_input_storage> meshes;
898};
899
900template <typename T>
901 requires(avnd::geometry_input_introspection<T>::size > 0)
902struct geometry_inputs_storage<T>
903{
904 // FIXME in Gfx/Graph/NodeRenderer.hpp
905 static_assert(avnd::geometry_input_introspection<T>::size == 1);
906
907 geometry_input_storage inputs[avnd::geometry_input_introspection<T>::size];
908 ossia::small_vector<QRhiBuffer*, 4> allocated;
909
910 struct pending_transform
911 {
912 ossia::transform3d value;
913 bool dirty{};
914 };
915 pending_transform transforms[avnd::geometry_input_introspection<T>::size];
916
917 void setTransform(int32_t port, const ossia::transform3d& v, auto& state)
918 {
919 avnd::geometry_input_introspection<T>::for_all_n2(
920 avnd::get_inputs<T>(state),
921 [&]<typename Field, std::size_t N, std::size_t NField>(
922 Field&, avnd::predicate_index<N>, avnd::field_index<NField>) {
923 if(int32_t(NField) == port)
924 transforms[N] = {v, true};
925 });
926 }
927
928 void readInputGeometries(
929 score::gfx::RenderList& renderer, const ossia::geometry_spec& spec, auto& parent,
930 auto& state)
931 {
932 // Copy the readback output inside the structure
933 // TODO it would be much better to do this inside the readback's
934 // "completed" callback.
935 avnd::geometry_input_introspection<T>::for_all_n(
936 avnd::get_inputs<T>(state),
937 [&]<typename Field, std::size_t N>(Field& t, avnd::predicate_index<N> np) {
938 this->inputs[N].spec = spec; // FIXME multiple geometry input ports
939 this->inputs[N].meshes.resize(1); // FIXME
940
941 // Here we fetch the readbacks results
942 auto& meshes = this->inputs[N].meshes[0];
943
944 oscr::meshes_from_ossia(
945 spec.meshes, t.mesh,
946 [&](auto& write_buf, int buffer_index, void* data, int64_t bytesize) {
947 // CPU input geometry, upload was done before
948 SCORE_ASSERT(buffer_index >= 0);
949 if(buffer_index < meshes.readbacks.size())
950 {
951 QRhiBuffer* handle = meshes.buffers[buffer_index];
952 write_buf.handle = handle;
953 write_buf.byte_size = handle->size();
954 }
955 }, [&](auto& write_buf, int buffer_index, void* handle) {
956 // GPU input buffer, CPU output buffer: need to fetch our readback
957 SCORE_ASSERT(buffer_index >= 0);
958 if(buffer_index < meshes.readbacks.size())
959 {
960 // FIXME investigate why runInitialPasses is called before inputAboutToFinish
961 auto& readback = meshes.readbacks[buffer_index].data;
962 write_buf.raw_data = reinterpret_cast<unsigned char*>(readback.data());
963 write_buf.byte_size = readback.size();
964 }
965 });
966
967 if constexpr(requires {
968 t.transform[0];
969 t.dirty_transform;
970 })
971 {
972 auto& pending = transforms[N];
973 if(pending.dirty)
974 {
975 static_assert(std::extent_v<std::remove_cvref_t<decltype(t.transform)>> == 16);
976 std::copy_n(pending.value.matrix, 16, t.transform);
977 pending.dirty = false;
978 t.dirty_transform = true;
979 }
980 else
981 {
982 t.dirty_transform = false;
983 }
984 }
985
986 if constexpr(requires { t.mesh.buffers[0].handle; })
987 {
988 if(spec.meshes && !spec.meshes->meshes.empty())
989 {
990 const auto& src = spec.meshes->meshes[0].buffers;
991 const std::size_t n
992 = std::min({src.size(), t.mesh.buffers.size(), meshes.buffers.size()});
993 for(std::size_t i = 0; i < n; i++)
994 {
995 if(!ossia::get_if<ossia::geometry::cpu_buffer>(&src[i].data))
996 continue;
997 if(QRhiBuffer* handle = meshes.buffers[i])
998 {
999 t.mesh.buffers[i].handle = handle;
1000 t.mesh.buffers[i].byte_size = handle->size();
1001 }
1002 }
1003 }
1004 }
1005 });
1006 }
1007
1008 void inputAboutToFinish(
1009 score::gfx::RenderList& renderer, QRhiResourceUpdateBatch*& res,
1010 const ossia::geometry_spec& spec, auto& state, auto& parent)
1011 {
1012 avnd::geometry_input_introspection<T>::for_all_n2(
1013 avnd::get_inputs<T>(state),
1014 [&]<typename Field, std::size_t N, std::size_t NField>(
1015 Field& t, avnd::predicate_index<N> np, avnd::field_index<NField> nf) {
1016 this->inputs[N].spec = spec; // FIXME multiple geometry input ports
1017 this->inputs[N].meshes.resize(1); // FIXME
1018 // Here we request readbacks if necessary
1019
1020 auto& meshes = this->inputs[N].meshes[0];
1021 auto upload = [&](int buffer_index, const void* data, int64_t bytesize) {
1022 if(meshes.buffers.size() <= std::size_t(buffer_index))
1023 {
1024 meshes.buffers.resize(buffer_index + 1);
1025 meshes.readbacks.resize(buffer_index + 1);
1026 }
1027 QRhiBuffer*& buf = meshes.buffers[buffer_index];
1028 if(!buf || !ossia::contains(allocated, buf))
1029 {
1030 buf = renderer.state.rhi->newBuffer(
1031 QRhiBuffer::Static,
1033 *renderer.state.rhi,
1034 QRhiBuffer::StorageBuffer | QRhiBuffer::VertexBuffer),
1035 bytesize);
1036 buf->setName(oscr::getUtf8Name<T>() + "::" + oscr::getUtf8Name(t));
1037 buf->create();
1038 allocated.push_back(buf);
1039 }
1040 else if(buf->size() < bytesize)
1041 {
1042 // Buffer exists but is too small — resize it.
1043 buf->setSize(bytesize);
1044 buf->create();
1045 }
1046
1047 res->uploadStaticBuffer(buf, 0, bytesize, data);
1048 };
1049 oscr::meshes_from_ossia(
1050 spec.meshes, t.mesh,
1051 [&](auto& write_buf, int buffer_index, void* data, int64_t bytesize) {
1052 // cpu -> gpu
1053 upload(buffer_index, data, bytesize);
1054 }, [&](auto& write_buf, int buffer_index, void* handle) {
1055 // gpu -> cpu
1056 if(meshes.readbacks.size() <= buffer_index)
1057 {
1058 meshes.buffers.resize(buffer_index + 1);
1059 meshes.readbacks.resize(buffer_index + 1);
1060 }
1061
1062 meshes.readbacks[buffer_index] = {};
1063 if(auto buf = static_cast<QRhiBuffer*>(handle))
1064 {
1065 meshes.buffers[buffer_index] = buf;
1066 res->readBackBuffer(buf, 0, buf->size(), &meshes.readbacks[buffer_index]);
1067 }
1068 else
1069 {
1070 meshes.buffers[buffer_index] = {};
1071 meshes.readbacks[buffer_index] = {};
1072 }
1073 });
1074
1075 // A CPU buffer is flagged dirty only on the frame it changed: a consumer
1076 // that first sees it later (inserted into a running graph, or rebuilt)
1077 // still has to upload it once.
1078 if constexpr(requires { t.mesh.buffers[0].handle; })
1079 {
1080 if(spec.meshes && !spec.meshes->meshes.empty())
1081 {
1082 const auto& src = spec.meshes->meshes[0].buffers;
1083 for(std::size_t i = 0; i < src.size(); i++)
1084 {
1085 auto* cpu = ossia::get_if<ossia::geometry::cpu_buffer>(&src[i].data);
1086 if(!cpu || !cpu->raw_data || cpu->byte_size <= 0)
1087 continue;
1088 if(i < meshes.buffers.size() && meshes.buffers[i]
1089 && ossia::contains(allocated, meshes.buffers[i]))
1090 continue;
1091 upload(int(i), cpu->raw_data.get(), cpu->byte_size);
1092 }
1093 }
1094 }
1095 });
1096 }
1097
1098 void release(score::gfx::RenderList& renderer)
1099 {
1100 for(auto& buf : allocated)
1101 renderer.releaseBuffer(buf);
1102 allocated.clear();
1103 for(auto& in : inputs)
1104 in.meshes.clear();
1105 }
1106};
1107
1108template <typename T>
1109 requires(avnd::geometry_input_introspection<T>::size == 0)
1110struct geometry_inputs_storage<T>
1111{
1112 static void readInputGeometries(auto&&...) { }
1113
1114 static void inputAboutToFinish(auto&&...) { }
1115
1116 static void release(auto&&...) { }
1117};
1118
1119template<typename T>
1120struct buffer_inputs_storage;
1121
1122template<typename T>
1123 requires (avnd::buffer_input_introspection<T>::size > 0)
1124struct buffer_inputs_storage<T>
1125{
1126 // +1 because of zero-array-size unsupported
1127 QRhiBufferReadbackResult
1128 m_readbacks[avnd::cpu_buffer_input_introspection<T>::size + 1];
1129 score::gfx::BufferView m_gpubufs[avnd::gpu_buffer_input_introspection<T>::size + 1];
1130
1131 void readInputBuffers(
1132 score::gfx::RenderList& renderer, auto& parent, auto& state)
1133 {
1134 if constexpr(avnd::cpu_buffer_input_introspection<T>::size > 0)
1135 {
1136 // Copy the readback output inside the structure
1137 // TODO it would be much better to do this inside the readback's
1138 // "completed" callback.
1139 avnd::cpu_buffer_input_introspection<T>::for_all_n(
1140 avnd::get_inputs<T>(state),
1141 [&]<typename Field, std::size_t N>
1142 (Field& t, avnd::predicate_index<N> np)
1143 {
1144 auto& readback = m_readbacks[N].data;
1145 t.buffer.raw_data = reinterpret_cast<unsigned char*>(readback.data());
1146 t.buffer.byte_size = readback.size();
1147 t.buffer.byte_offset = 0; // FIXME
1148 t.buffer.changed = true;
1149 });
1150 }
1151
1152 if constexpr(avnd::gpu_buffer_input_introspection<T>::size > 0)
1153 {
1154 // Copy the readback output inside the structure
1155 // TODO it would be much better to do this inside the readback's
1156 // "completed" callback.
1157 avnd::gpu_buffer_input_introspection<T>::for_all_n2(
1158 avnd::get_inputs<T>(state),
1159 [&]<typename Field, std::size_t N, std::size_t NField>(
1160 Field& t, avnd::predicate_index<N> np, avnd::field_index<NField> nf) {
1161 score::gfx::BufferView& buf = m_gpubufs[N];
1162 if(!buf)
1163 buf = getInputBuffer(renderer, parent, nf);
1164 if(!buf)
1165 return;
1166 t.buffer.handle = buf.handle;
1167 t.buffer.byte_size = buf.byte_size;
1168 t.buffer.byte_offset = buf.byte_offset;
1169 // t.buffer.changed = true; FIXME
1170 });
1171 }
1172 }
1173
1174 void inputAboutToFinish(
1175 score::gfx::RenderList& renderer,
1176 QRhiResourceUpdateBatch*& res,
1177 auto& state,
1178 auto& parent)
1179 {
1180 avnd::cpu_buffer_input_introspection<T>::for_all_n2(
1181 avnd::get_inputs<T>(state),
1182 [&]<typename Field, std::size_t N, std::size_t NField>
1183 (Field& port, avnd::predicate_index<N> np, avnd::field_index<NField> nf) {
1184 readbackInputBuffer(renderer, *res, parent, m_readbacks[N], nf);
1185 });
1186 avnd::gpu_buffer_input_introspection<T>::for_all_n2(
1187 avnd::get_inputs<T>(state),
1188 [&]<typename Field, std::size_t N, std::size_t NField>
1189 (Field& port, avnd::predicate_index<N> np, avnd::field_index<NField> nf) {
1190 m_gpubufs[N] = getInputBuffer(renderer, parent, nf);
1191 });
1192 }
1193};
1194
1195template<typename T>
1196 requires (avnd::buffer_input_introspection<T>::size == 0)
1197struct buffer_inputs_storage<T>
1198{
1199 static void readInputBuffers(auto&&...)
1200 {
1201
1202 }
1203
1204 static void inputAboutToFinish(auto&&...)
1205 {
1206
1207 }
1208};
1209
1210struct MaybeOwnedBuffer : score::gfx::BufferView
1211{
1212 bool owned{false};
1213};
1214
1215template<typename T>
1216struct buffer_outputs_storage;
1217
1218template<typename T>
1219 requires (avnd::buffer_output_introspection<T>::size > 0)
1220struct buffer_outputs_storage<T>
1221{
1222 std::pair<const score::gfx::Port*, MaybeOwnedBuffer>
1223 m_buffers[avnd::buffer_output_introspection<T>::size];
1224
1225 QRhiResourceUpdateBatch* currentResourceUpdateBatch{};
1226
1229 score::gfx::RenderList* m_renderer{};
1230
1231 template <typename Field, std::size_t N, std::size_t NField>
1232 requires avnd::cpu_buffer<std::decay_t<decltype(Field::buffer)>>
1233 void createOutput(
1234 score::gfx::RenderList& renderer, auto& parent, Field& port,
1235 avnd::predicate_index<N> np, avnd::field_index<NField> nf)
1236 {
1237 auto& [gfx_port, buf] = m_buffers[N];
1238 gfx_port = parent.output[nf];
1239 buf.handle = renderer.state.rhi->newBuffer(
1240 QRhiBuffer::Static,
1242 *renderer.state.rhi,
1243 QRhiBuffer::StorageBuffer | QRhiBuffer::VertexBuffer), 1);
1244 buf.handle->setName(oscr::getUtf8Name<T>() + "::" + oscr::getUtf8Name(port));
1245 buf.byte_offset = 0;
1246 buf.byte_size = 1;
1247 buf.owned = true;
1248
1249 buf.handle->create();
1250
1251 m_renderer = &renderer;
1252 bindUpload(port, np);
1253 }
1254
1262 template <typename Field, std::size_t N>
1263 void bindUpload(Field& port, avnd::predicate_index<N>)
1264 {
1265 port.buffer.upload
1266 = [this, &port](const char* data, int64_t offset, int64_t bytesize) {
1267 // FIXME is offset and bytesize relative to the input or the output data ?
1268 SCORE_ASSERT(currentResourceUpdateBatch);
1269 SCORE_ASSERT(m_renderer);
1270 auto& rhi = *m_renderer->state.rhi;
1271 auto& [gfx_port, buf] = m_buffers[N];
1272
1273 if(!buf.handle)
1274 {
1275 if(bytesize > 0)
1276 {
1277 buf.handle = rhi.newBuffer(
1278 QRhiBuffer::Static,
1280 rhi, QRhiBuffer::StorageBuffer | QRhiBuffer::VertexBuffer),
1281 bytesize);
1282 buf.handle->setName(oscr::getUtf8Name<T>() + "::" + oscr::getUtf8Name(port));
1283 buf.byte_offset = 0;
1284 buf.byte_size = bytesize;
1285 buf.owned = true;
1286
1287 buf.handle->create();
1288 }
1289 else
1290 {
1291 buf.handle = rhi.newBuffer(
1292 QRhiBuffer::Static,
1294 rhi, QRhiBuffer::StorageBuffer | QRhiBuffer::VertexBuffer),
1295 1);
1296 buf.handle->setName(oscr::getUtf8Name<T>() + "::" + oscr::getUtf8Name(port));
1297 buf.byte_offset = 0;
1298 buf.byte_size = 1;
1299 buf.owned = true;
1300
1301 buf.handle->create();
1302 return;
1303 }
1304 }
1305 else if(buf.handle->size() != bytesize)
1306 {
1307 buf.handle->destroy();
1308 buf.handle->setSize(bytesize);
1309 buf.handle->create();
1310 buf.byte_size = bytesize;
1311 }
1312
1314 currentResourceUpdateBatch, buf.handle, offset, bytesize, data);
1315 };
1316 }
1317
1320 void bindUploads(auto& state)
1321 {
1322 avnd::buffer_output_introspection<T>::for_all_n(
1323 avnd::get_outputs<T>(state),
1324 [this]<typename Field, std::size_t N>(Field& port, avnd::predicate_index<N> np) {
1325 if constexpr(avnd::cpu_buffer<std::decay_t<decltype(Field::buffer)>> && requires {
1326 port.buffer.upload(nullptr, 0, 0);
1327 })
1328 {
1329 bindUpload(port, np);
1330 }
1331 });
1332 }
1333
1334 template <typename Field, std::size_t N, std::size_t NField>
1335 requires avnd::gpu_buffer<std::decay_t<decltype(Field::buffer)>>
1336 void createOutput(
1337 score::gfx::RenderList& renderer, auto& parent, Field& port,
1338 avnd::predicate_index<N> np, avnd::field_index<NField> nf)
1339 {
1340 auto& [gfx_port, buf] = m_buffers[N];
1341 gfx_port = parent.output[nf];
1342 buf.handle = reinterpret_cast<QRhiBuffer*>(port.buffer.handle);
1343 buf.byte_size = port.buffer.byte_size;
1344 buf.byte_offset = port.buffer.byte_offset;
1345 buf.owned = false;
1346 }
1347
1348 void init(score::gfx::RenderList& renderer, auto& state, auto& parent)
1349 {
1350 // Init buffers for the outputs
1351 avnd::buffer_output_introspection<T>::for_all_n2(
1352 avnd::get_outputs<T>(state), [&]<typename Field, std::size_t N, std::size_t NField>
1353 (Field& port, avnd::predicate_index<N> np, avnd::field_index<NField> nf) {
1354 SCORE_ASSERT(parent.output.size() > nf);
1355 SCORE_ASSERT(parent.output[nf]->type == score::gfx::Types::Buffer);
1356 using buffer_type = std::decay_t<decltype(port.buffer)>;
1357
1358 if constexpr(avnd::cpu_raw_buffer<buffer_type> && requires {
1359 port.buffer.upload(nullptr, 0, 0);
1360 })
1361 {
1362 createOutput(renderer, parent, port, np, nf);
1363 }
1364 else if constexpr(avnd::gpu_buffer<buffer_type>)
1365 {
1366 createOutput(renderer, parent, port, np, nf);
1367 }
1368 else
1369 {
1370 // m_buffers[N] = createOutput(renderer, *parent.output[nf], port.buffer);
1371 static_assert(std::is_same_v<T, void>, "unsupported");
1372 }
1373 });
1374 }
1375
1376 void prepareUpload(QRhiResourceUpdateBatch& res)
1377 {
1378 currentResourceUpdateBatch = &res;
1379 }
1380
1381 void upload(score::gfx::RenderList& renderer, auto& state, QRhiResourceUpdateBatch& res)
1382 {
1383 avnd::buffer_output_introspection<T>::for_all_n(
1384 avnd::get_outputs<T>(state), [&]<std::size_t N>(auto& t, avnd::predicate_index<N> idx) {
1385 auto& [port, buf] = m_buffers[N];
1386 uploadOutputBuffer(renderer, t.buffer, res, buf);
1387 });
1388 }
1389
1390 void release(score::gfx::RenderList& renderer)
1391 {
1392 // Free outputs
1393 for(auto& [p, buf] : m_buffers)
1394 {
1395 if(buf.owned)
1396 renderer.releaseBuffer(buf.handle);
1397 buf.handle = nullptr;
1398 buf.owned = false;
1399 }
1400 }
1401};
1402
1403template<typename T>
1404 requires (avnd::buffer_output_introspection<T>::size == 0)
1405struct buffer_outputs_storage<T>
1406{
1407 static void init(auto&&...)
1408 {
1409
1410 }
1411
1412 static void prepareUpload(auto&&...)
1413 {
1414 }
1415
1416 static void bindUploads(auto&&...)
1417 {
1418 }
1419
1420 static void upload(auto&&...)
1421 {
1422 }
1423
1424 static void release(auto&&...)
1425 {
1426 }
1427};
1428
1429
1430template <typename Tex>
1431static auto
1432createOutputTexture(score::gfx::RenderList& renderer, const Tex& texture_spec, QSize size)
1433{
1434 auto& rhi = *renderer.state.rhi;
1435 QRhiTexture* texture = &renderer.emptyTexture();
1436 if(size.width() > 0 && size.height() > 0)
1437 {
1438 texture = rhi.newTexture(
1439 gpp::qrhi::textureFormat(texture_spec), size, 1, QRhiTexture::Flag{});
1440
1441 texture->create();
1442 }
1443
1444 auto sampler = rhi.newSampler(
1445 QRhiSampler::Linear, QRhiSampler::Linear, QRhiSampler::None,
1446 QRhiSampler::ClampToEdge, QRhiSampler::ClampToEdge);
1447
1448 sampler->create();
1449 return score::gfx::Sampler{sampler, texture};
1450}
1451
1452
1453template<typename T>
1454struct texture_inputs_storage;
1455
1456template<typename T>
1457 requires (avnd::texture_input_introspection<T>::size > 0)
1458struct texture_inputs_storage<T>
1459{
1460 ossia::small_flat_map<const score::gfx::Port*, score::gfx::TextureRenderTarget, 2>
1461 m_rts;
1462 ossia::small_flat_map<const score::gfx::Port*, QRhiSampler*, 2> m_samplers;
1463
1464 QRhiReadbackResult m_readbacks[avnd::texture_input_introspection<T>::size];
1465 bool m_readbackPremultiplied[avnd::texture_input_introspection<T>::size]{};
1466
1467 template <typename Tex>
1468 QRhiTexture* createInput(
1469 score::gfx::RenderList& renderer, score::gfx::Port* port, Tex& texture_spec,
1470 const score::gfx::RenderTargetSpecs& spec, bool wantsSamplableDepth = false,
1471 bool mipmapped = false)
1472 {
1473 QRhiTexture::Flags flags
1474 = QRhiTexture::RenderTarget | QRhiTexture::UsedAsTransferSource;
1475 if(mipmapped)
1476 flags |= QRhiTexture::MipMapped | QRhiTexture::UsedWithGenerateMips;
1477 QRhiTexture::Format fmt{};
1478 if constexpr(requires (Tex tex) { tex.format = {}; } && !requires (Tex tex) { tex.request_format; })
1479 {
1480 // Format freely assignable: we use what the user sets in the GUI
1481 fmt = spec.format;
1482 gpp::qrhi::toTextureFormat(fmt, texture_spec);
1483 }
1484 else
1485 {
1486 fmt = gpp::qrhi::textureFormat(texture_spec);
1487 }
1488
1489 QRhiTexture* texture = renderer.state.rhi->newTexture(
1490 fmt, spec.size, 1, flags);
1491
1492 SCORE_ASSERT(texture->create());
1493 // wantsSamplableDepth implies wantsDepth: createRenderTarget allocates
1494 // a sampleable single-sample depth texture (with MSAA-resolve when
1495 // available) instead of a renderbuffer / non-resolve depth target.
1496 // Same shape ISF/CSF inputs get when their port has SamplableDepth.
1497 const bool wantsDepth = renderer.requiresDepth(*port) || wantsSamplableDepth;
1498 m_rts[port] = score::gfx::createRenderTarget(
1499 renderer.state, texture, renderer.samples(),
1500 wantsDepth, wantsSamplableDepth);
1501 return texture;
1502 }
1503
1504 void init(auto& self, score::gfx::RenderList& renderer)
1505 {
1506 // Init input render targets
1507 avnd::texture_input_introspection<T>::for_all_n2(
1508 avnd::get_inputs<T>(*self.state),
1509 [&]<typename F, std::size_t K, std::size_t N>(F& t, avnd::predicate_index<K>, avnd::field_index<N>) {
1510 // Non-2D GPU texture inputs (cube / array / 3D) don't get a local
1511 // render target — the port carries Flag::GrabsFromSource (set by
1512 // initGfxPorts via texture_kind_of<F>()), the graph will populate
1513 // t.texture.handle through updateInputTexture when the edge
1514 // resolves. Skipping the allocation here avoids wasting a 2D
1515 // colour attachment that would never be rendered into anyway.
1516 if constexpr(avnd::gpu_texture_port<F>
1517 && halp::texture_kind_of<F>() != halp::texture_kind::texture_2d)
1518 {
1519 t.texture.kind = halp::texture_kind_of<F>();
1520 // Handle + size populated later by updateInputTexture once the
1521 // upstream is resolved.
1522 return;
1523 }
1524
1525 auto& parent = self.node();
1526 auto spec = renderer.resolveInputRenderTargetSpecs(parent, N);
1527 if constexpr(requires {
1528 t.request_width;
1529 t.request_height;
1530 })
1531 {
1532 spec.size.rwidth() = t.request_width;
1533 spec.size.rheight() = t.request_height;
1534 }
1535
1536 constexpr bool wantsSamplableDepth
1537 = avnd::gpu_texture_port<F> && halp::samplable_depth_of<F>();
1538 if constexpr(avnd::gpu_texture_port<F>)
1539 {
1540 if(!single_cable_port<F> || singleCableNeedsTarget(renderer, *parent.input[N]))
1541 createInput(
1542 renderer, parent.input[N], t.texture, spec, wantsSamplableDepth,
1543 spec.mipmap_mode != QRhiSampler::None);
1544 auto* sampler = renderer.state.rhi->newSampler(
1545 spec.mag_filter, spec.min_filter, spec.mipmap_mode, spec.address_u,
1546 spec.address_v, spec.address_w);
1547 sampler->setName("texture_inputs_storage::sampler");
1548 sampler->create();
1549 m_samplers[parent.input[N]] = sampler;
1550 }
1551 else
1552 {
1553 createInput(renderer, parent.input[N], t.texture, spec, wantsSamplableDepth);
1554 }
1555 if constexpr(avnd::cpu_texture_port<F>)
1556 {
1557 t.texture.width = spec.size.width();
1558 t.texture.height = spec.size.height();
1559 }
1560 });
1561
1562 refreshGpuInputs(self, renderer);
1563 }
1564
1565 static std::pair<bool, QRhiTexture*> upstreamTexture(
1566 score::gfx::RenderList& renderer, const score::gfx::Node& node, int32_t index,
1567 const score::gfx::Port& port, bool needsOwnTarget = false)
1568 {
1569 int wired = 0;
1570 QRhiTexture* direct = nullptr;
1571 for(auto* edge : port.edges)
1572 {
1573 if(!edge || !edge->source || !edge->source->node)
1574 continue;
1575 auto& rendered = edge->source->node->renderedNodes;
1576 auto it = rendered.find(&renderer);
1577 if(it == rendered.end() || !it->second)
1578 continue;
1579 if(wired++ == 0)
1580 direct = it->second->textureForOutput(*edge->source);
1581 }
1582 if(wired != 1 || needsOwnTarget || node.hasExplicitRenderTargetSpecs(index))
1583 direct = nullptr;
1584 return {wired > 0, direct};
1585 }
1586
1587 static QRhiTexture*
1588 singleCableTexture(score::gfx::RenderList& renderer, const score::gfx::Port& port)
1589 {
1590 for(auto* edge : port.edges)
1591 {
1592 if(!edge || !edge->source || !edge->source->node)
1593 continue;
1594 auto& rendered = edge->source->node->renderedNodes;
1595 auto it = rendered.find(&renderer);
1596 if(it == rendered.end() || !it->second)
1597 continue;
1598 return it->second->textureForOutput(*edge->source);
1599 }
1600 return nullptr;
1601 }
1602
1603 static bool
1604 singleCableNeedsTarget(score::gfx::RenderList& renderer, const score::gfx::Port& port)
1605 {
1606 return !port.edges.empty() && !singleCableTexture(renderer, port);
1607 }
1608
1609 static bool
1610 composited(score::gfx::RenderList& renderer, const score::gfx::Port& port)
1611 {
1612 for(auto* edge : port.edges)
1613 {
1614 if(!edge || !edge->source || !edge->source->node)
1615 continue;
1616 auto& rendered = edge->source->node->renderedNodes;
1617 if(auto it = rendered.find(&renderer);
1618 it != rendered.end() && it->second && it->second->hasOutputPassForEdge(*edge))
1619 return true;
1620 }
1621 return false;
1622 }
1623
1624 void releaseTarget(const score::gfx::Port* port)
1625 {
1626 if(auto it = m_rts.find(port); it != m_rts.end())
1627 {
1628 it->second.release();
1629 m_rts.erase(it);
1630 }
1631 }
1632
1633 QRhiSampler* refreshSampler(
1634 const score::gfx::Node& node, int32_t index, score::gfx::RenderList& renderer,
1635 const score::gfx::Port* port)
1636 {
1637 auto it = m_samplers.find(port);
1638 if(it == m_samplers.end() || !it->second)
1639 return nullptr;
1640 auto* sampler = it->second;
1641 const auto spec = node.resolveRenderTargetSpecs(index, renderer);
1642 if(sampler->magFilter() != spec.mag_filter || sampler->minFilter() != spec.min_filter
1643 || sampler->mipmapMode() != spec.mipmap_mode
1644 || sampler->addressU() != spec.address_u || sampler->addressV() != spec.address_v
1645 || sampler->addressW() != spec.address_w)
1646 {
1647 sampler->destroy();
1648 sampler->setMagFilter(spec.mag_filter);
1649 sampler->setMinFilter(spec.min_filter);
1650 sampler->setMipmapMode(spec.mipmap_mode);
1651 sampler->setAddressU(spec.address_u);
1652 sampler->setAddressV(spec.address_v);
1653 sampler->setAddressW(spec.address_w);
1654 sampler->create();
1655 }
1656 return sampler;
1657 }
1658
1659 static void describeTexture(halp::gpu_texture& dst, QRhiTexture& tex)
1660 {
1661 const auto sz = tex.pixelSize();
1662 dst.handle = &tex;
1663 dst.width = sz.width();
1664 dst.height = sz.height();
1665 gpp::qrhi::toTextureFormat(tex.format(), dst);
1666 const auto flags = tex.flags();
1667 if(flags & QRhiTexture::CubeMap)
1668 {
1669 dst.kind = halp::texture_kind::cubemap;
1670 dst.layers_or_depth = 6;
1671 }
1672 else if(flags & QRhiTexture::ThreeDimensional)
1673 {
1674 dst.kind = halp::texture_kind::texture_3d;
1675 dst.layers_or_depth = tex.depth();
1676 }
1677 else if(flags & QRhiTexture::TextureArray)
1678 {
1679 dst.kind = halp::texture_kind::texture_array;
1680 dst.layers_or_depth = tex.arraySize();
1681 }
1682 else
1683 {
1684 dst.kind = halp::texture_kind::texture_2d;
1685 dst.layers_or_depth = 1;
1686 }
1687 }
1688
1689 void refreshGpuInputs(auto& self, score::gfx::RenderList& renderer)
1690 {
1691 avnd::texture_input_introspection<T>::for_all_n2(
1692 avnd::get_inputs<T>(*self.state),
1693 [&]<typename F, std::size_t K, std::size_t N>(
1694 F& t, avnd::predicate_index<K>, avnd::field_index<N>) {
1695 if constexpr(
1696 avnd::gpu_texture_port<F>
1697 && halp::texture_kind_of<F>() == halp::texture_kind::texture_2d)
1698 {
1699 constexpr bool wantsSamplableDepth = halp::samplable_depth_of<F>();
1700 auto* port = self.node().input[N];
1701 auto& tex = t.texture;
1702 const auto rt_it = m_rts.find(port);
1703 const bool mipmapped
1704 = rt_it != m_rts.end() && rt_it->second.texture
1705 && rt_it->second.texture->flags().testFlag(QRhiTexture::MipMapped);
1706 tex.sampler_handle = refreshSampler(self.node(), N, renderer, port);
1707
1708 QRhiTexture* src = nullptr;
1709 if constexpr(single_cable_port<F>)
1710 {
1711 src = singleCableTexture(renderer, *port);
1712 if(src)
1713 {
1714 if(rt_it != m_rts.end() && !composited(renderer, *port))
1715 releaseTarget(port);
1716 }
1717 else if(!port->edges.empty() && rt_it != m_rts.end())
1718 {
1719 src = rt_it->second.texture;
1720 }
1721 }
1722 else if(auto [wired, direct]
1723 = upstreamTexture(renderer, self.node(), N, *port, mipmapped);
1724 wired)
1725 {
1726 if constexpr(!wantsSamplableDepth)
1727 src = direct;
1728 if(!src && rt_it != m_rts.end())
1729 src = rt_it->second.texture;
1730 }
1731
1732 if(!src)
1733 {
1734 tex.handle = nullptr;
1735 tex.width = 0;
1736 tex.height = 0;
1737 tex.kind = halp::texture_kind::texture_2d;
1738 tex.layers_or_depth = 1;
1739 if constexpr(wantsSamplableDepth)
1740 tex.depth_handle = nullptr;
1741 return;
1742 }
1743
1744 describeTexture(tex, *src);
1745 if constexpr(wantsSamplableDepth)
1746 {
1747 auto* depth = rt_it->second.depthTexture;
1748 tex.depth_handle = depth;
1749 if(depth)
1750 tex.depth_format = qrhiToHalpDepthFormat(depth->format());
1751 }
1752 }
1753 });
1754 }
1755
1756 bool update(auto& self,
1757 score::gfx::RenderList& renderer, QRhiResourceUpdateBatch& res)
1758 {
1759 avnd::texture_input_introspection<T>::for_all_n2(
1760 avnd::get_inputs<T>(*self.state),
1761 [&]<typename F, std::size_t K, std::size_t N>(
1762 F& t, avnd::predicate_index<K>, avnd::field_index<N>) {
1763 if constexpr(single_cable_port<F> && avnd::gpu_texture_port<F>)
1764 {
1765 auto& parent = self.node();
1766 auto* port = parent.input[N];
1767 if(m_rts.find(port) != m_rts.end() || !singleCableNeedsTarget(renderer, *port))
1768 return;
1769 auto spec = renderer.resolveInputRenderTargetSpecs(parent, N);
1770 if constexpr(requires {
1771 t.request_width;
1772 t.request_height;
1773 })
1774 {
1775 spec.size.rwidth() = t.request_width;
1776 spec.size.rheight() = t.request_height;
1777 }
1778 createInput(
1779 renderer, port, t.texture, spec, false,
1780 spec.mipmap_mode != QRhiSampler::None);
1781 for(auto* edge : port->edges)
1782 {
1783 auto& rendered = edge->source->node->renderedNodes;
1784 if(auto it = rendered.find(&renderer); it != rendered.end() && it->second)
1785 {
1786 it->second->removeOutputPass(renderer, *edge);
1787 it->second->addOutputPass(renderer, *edge, res);
1788 }
1789 }
1790 }
1791 });
1792#if 0
1793 bool need_update = false;
1794 avnd::texture_input_introspection<T>::for_all_n2(
1795 avnd::get_inputs<T>(*self.state),
1796 [&]<typename F, std::size_t K, std::size_t N>(F& t, avnd::predicate_index<K>, avnd::field_index<N>) {
1797 if constexpr(requires {
1798 t.request_width;
1799 t.request_height;
1800 })
1801 {
1802 auto& parent = self.node();
1803 auto port = parent.input[N];
1804 const score::gfx::TextureRenderTarget& texture = m_rts[port];
1805 QSizeF sz{};
1806 if(texture.texture)
1807 sz = texture.texture->pixelSize();
1808 if(sz.width() != t.request_width || sz.height() != t.request_height)
1809 {
1810 // FIXME right now this doesn't work because
1811 // the render target spec is stored in the node.
1812 // Also the RenderList just recomputes everything anyways,
1813 // so we should just emit a "need to change" signal and abort as
1814 // long as things aren't more optimized and actually follow the graph
1815
1816 // m_rts[port].release();
1817
1818 // auto spec = parent.resolveRenderTargetSpecs(N, renderer);
1819 // spec.size.rwidth() = t.request_width;
1820 // spec.size.rheight() = t.request_height;
1821
1822 // createInput(renderer, port, t.texture, sz);
1823
1824 // t.texture.width = spec.size.width();
1825 // t.texture.height = spec.size.height();
1826 // need_update = true;
1827 //
1828 }
1829 }
1830 });
1831 return need_update;
1832#endif
1833 return false;
1834 }
1835
1836 template <typename F>
1837 static constexpr bool reads_upstream_directly() noexcept
1838 {
1839 using Tex = std::decay_t<decltype(F::texture)>;
1840 return avnd::cpu_texture_port<F>
1841 && requires(Tex tex) { tex.format = {}; }
1842 && !requires(Tex tex) { tex.request_format; }
1843 && !requires(F f) { f.request_width; };
1844 }
1845
1846 void runInitialPasses(auto& self, score::gfx::RenderList& renderer)
1847 {
1848 auto& rhi = *renderer.state.rhi;
1849 refreshGpuInputs(self, renderer);
1850
1851 // Fetch input textures (if any)
1852 // Copy the readback output inside the structure
1853 // TODO it would be much better to do this inside the readback's
1854 // "completed" callback.
1855 if constexpr(avnd::cpu_texture_input_introspection<T>::size > 0)
1856 {
1857 avnd::texture_input_introspection<T>::for_all_n(
1858 avnd::get_inputs<T>(*self.state), [&]<typename F, std::size_t K>(F& t, avnd::predicate_index<K>) {
1859 if constexpr(avnd::cpu_texture_port<F>)
1860 {
1861 if constexpr(reads_upstream_directly<F>())
1862 {
1863 const auto& rb = m_readbacks[K];
1864 if(!rb.pixelSize.isEmpty())
1865 {
1866 t.texture.width = rb.pixelSize.width();
1867 t.texture.height = rb.pixelSize.height();
1868 gpp::qrhi::toTextureFormat(rb.format, t.texture);
1869 }
1870 }
1871 if(auto& rb = m_readbacks[K]; m_readbackPremultiplied[K] && !rb.data.isEmpty())
1872 {
1873 gpp::qrhi::unpremultiply(
1874 rb.format, rb.data, rb.pixelSize.width(), rb.pixelSize.height());
1875 m_readbackPremultiplied[K] = false;
1876 }
1877 oscr::loadInputTexture(rhi, m_readbacks, t.texture, K);
1878 }
1879 });
1880 }
1881 }
1882
1883 void release()
1884 {
1885 // Free inputs
1886 // TODO investigate why reference does not work here:
1887 for(auto [port, rt] : m_rts)
1888 rt.release();
1889 m_rts.clear();
1890 for(auto [port, sampler] : m_samplers)
1891 sampler->deleteLater();
1892 m_samplers.clear();
1893 }
1894
1895 void inputAboutToFinish(
1896 auto& self, score::gfx::RenderList& renderer, const score::gfx::Port& p,
1897 QRhiResourceUpdateBatch*& res)
1898 {
1899 if constexpr(avnd::cpu_texture_input_introspection<T>::size > 0)
1900 {
1901 const auto& inputs = self.node().input;
1902 avnd::texture_input_introspection<T>::for_all_n2(
1903 avnd::get_inputs<T>(*self.state),
1904 [&]<typename F, std::size_t K, std::size_t N>(
1905 F&, avnd::predicate_index<K>, avnd::field_index<N>) {
1906 if constexpr(avnd::cpu_texture_port<F>)
1907 {
1908 if(N >= inputs.size() || inputs[N] != &p)
1909 return;
1910 auto rt_it = m_rts.find(&p);
1911 if(rt_it == m_rts.end())
1912 return;
1913 QRhiTexture* tex = rt_it->second.texture;
1914 if constexpr(reads_upstream_directly<F>())
1915 {
1916 auto [wired, direct] = upstreamTexture(renderer, self.node(), N, p);
1917 if(direct && direct->sampleCount() <= 1
1918 && (direct->flags() & QRhiTexture::UsedAsTransferSource)
1919 && !(direct->flags()
1920 & (QRhiTexture::CubeMap | QRhiTexture::ThreeDimensional
1921 | QRhiTexture::TextureArray)))
1922 tex = direct;
1923 }
1924 if(!tex)
1925 return;
1926 auto& readback = m_readbacks[K];
1927 readback = {};
1928 m_readbackPremultiplied[K] = tex == rt_it->second.texture;
1929 res->readBackTexture(QRhiReadbackDescription{tex}, &readback);
1930 }
1931 });
1932 }
1933 }
1934
1935};
1936template<typename T>
1937 requires (avnd::texture_input_introspection<T>::size == 0)
1938struct texture_inputs_storage<T>
1939{
1940 static void init(auto&&...) { }
1941 static void runInitialPasses(auto&&...) { }
1942 static void release(auto&&...) { }
1943 static void inputAboutToFinish(auto&&...) { }
1944};
1945
1946
1947
1948template <avnd::cpu_texture Tex>
1949static QRhiTexture* updateTexture(auto& self, score::gfx::RenderList& renderer, int k, const Tex& cpu_tex)
1950{
1951 auto& [sampler, texture, fb_] = self.m_samplers[k];
1952 if(texture)
1953 {
1954 auto sz = texture->pixelSize();
1955 if(cpu_tex.width == sz.width() && cpu_tex.height == sz.height())
1956 return texture;
1957 }
1958
1959 // Check the texture size
1960 if(cpu_tex.width > 0 && cpu_tex.height > 0)
1961 {
1962 QRhiTexture* oldtex = texture;
1963 QRhiTexture* newtex = renderer.state.rhi->newTexture(
1964 gpp::qrhi::textureFormat(cpu_tex), QSize{cpu_tex.width, cpu_tex.height}, 1,
1965 QRhiTexture::Flag{});
1966 newtex->create();
1967 for(auto& [edge, pass] : self.m_p)
1968 if(pass.p.srb)
1969 score::gfx::replaceTexture(*pass.p.srb, sampler, newtex);
1970 texture = newtex;
1971
1972 if(oldtex && oldtex != &renderer.emptyTexture())
1973 {
1974 oldtex->deleteLater();
1975 }
1976
1977 return newtex;
1978 }
1979 else
1980 {
1981 for(auto& [edge, pass] : self.m_p)
1982 if(pass.p.srb)
1983 score::gfx::replaceTexture(*pass.p.srb, sampler, &renderer.emptyTexture());
1984
1985 return &renderer.emptyTexture();
1986 }
1987}
1988
1989template <avnd::cpu_texture Tex>
1990static void uploadOutputTexture(auto& self,
1991 score::gfx::RenderList& renderer, int k, Tex& cpu_tex,
1992 QRhiResourceUpdateBatch* res)
1993{
1994 if(cpu_tex.changed)
1995 {
1996 if(auto texture = updateTexture(self, renderer, k, cpu_tex))
1997 {
1998 if(!cpu_tex.bytes || cpu_tex.bytesize() <= 0)
1999 {
2000 cpu_tex.changed = false;
2001 return;
2002 }
2003
2004 QByteArray buf
2005 = QByteArray::fromRawData((const char*)cpu_tex.bytes, cpu_tex.bytesize());
2006 if constexpr(requires { Tex::RGB; })
2007 {
2008 // RGB -> RGBA
2009 // FIXME other conversions
2010 const QByteArray rgb = buf;
2011 QByteArray rgba;
2012 rgba.resize(cpu_tex.width * cpu_tex.height * 4);
2013 auto src = (const unsigned char*)rgb.constData();
2014 auto dst = (unsigned char*)rgba.data();
2015 for(int rgb_byte = 0, rgba_byte = 0, N = rgb.size(); rgb_byte < N;)
2016 {
2017 dst[rgba_byte + 0] = src[rgb_byte + 0];
2018 dst[rgba_byte + 1] = src[rgb_byte + 1];
2019 dst[rgba_byte + 2] = src[rgb_byte + 2];
2020 dst[rgba_byte + 3] = 255;
2021 rgb_byte += 3;
2022 rgba_byte += 4;
2023 }
2024 buf = rgba;
2025 }
2026
2027 // Upload it (mirroring is done in shader generic_texgen_fs if necessary)
2028 {
2029 QRhiTextureSubresourceUploadDescription sd(buf);
2030 QRhiTextureUploadDescription desc{QRhiTextureUploadEntry{0, 0, sd}};
2031
2032 res->uploadTexture(texture, desc);
2033 }
2034
2035 cpu_tex.changed = false;
2036 }
2037 }
2038}
2039
2040static const constexpr auto generic_texgen_vs = R"_(#version 450
2041layout(location = 0) in vec2 position;
2042layout(location = 1) in vec2 texcoord;
2043
2044layout(binding=3) uniform sampler2D y_tex;
2045layout(location = 0) out vec2 v_texcoord;
2046
2047layout(std140, binding = 0) uniform renderer_t {
2048 mat4 clipSpaceCorrMatrix;
2049 vec2 renderSize;
2050} renderer;
2051
2052out gl_PerVertex { vec4 gl_Position; };
2053
2054void main()
2055{
2056#if defined(QSHADER_SPIRV) || defined(QSHADER_GLSL)
2057 v_texcoord = vec2(texcoord.x, 1. - texcoord.y);
2058#else
2059 v_texcoord = texcoord;
2060#endif
2061 gl_Position = renderer.clipSpaceCorrMatrix * vec4(position.xy, 0.0, 1.);
2062}
2063)_";
2064
2065static const constexpr auto generic_texgen_fs = R"_(#version 450
2066layout(location = 0) in vec2 v_texcoord;
2067layout(location = 0) out vec4 fragColor;
2068
2069layout(std140, binding = 0) uniform renderer_t {
2070mat4 clipSpaceCorrMatrix;
2071vec2 renderSize;
2072} renderer;
2073
2074layout(binding=3) uniform sampler2D y_tex;
2075
2076void main ()
2077{
2078 fragColor = texture(y_tex, v_texcoord);
2079}
2080)_";
2081
2082template<typename T>
2083struct texture_outputs_storage;
2084
2085// If we have texture outs we need the whole rendering infrastructure
2086template<typename T>
2087 requires (avnd::texture_output_introspection<T>::size > 0)
2088struct texture_outputs_storage<T>
2089{
2090 static bool drawnOutputPremultiplied(auto& self) noexcept
2091 {
2092 bool premultiplied = false;
2093 const auto first = [&]<typename F, std::size_t N>(F&, avnd::predicate_index<N>) {
2094 if constexpr(N == 0)
2095 premultiplied = requires { F::premultiplied; };
2096 };
2097 if constexpr(avnd::cpu_texture_output_introspection<T>::size > 0)
2098 avnd::cpu_texture_output_introspection<T>::for_all_n(
2099 avnd::get_outputs<T>(*self.state), first);
2100 else
2101 avnd::gpu_texture_output_introspection<T>::for_all_n(
2102 avnd::get_outputs<T>(*self.state), first);
2103 return premultiplied;
2104 }
2105
2106 void init(auto& self, score::gfx::RenderList& renderer, QRhiResourceUpdateBatch& res)
2107 {
2108 const auto& mesh = renderer.defaultTriangle();
2109 self.defaultMeshInit(renderer, mesh, res);
2110 self.processUBOInit(renderer);
2111 // Not needed here as we do not have a GPU pass:
2112 // this->m_material.init(renderer, this->node.input, this->m_samplers);
2113
2114 std::tie(self.m_vertexS, self.m_fragmentS)
2115 = score::gfx::makeShaders(renderer.state, generic_texgen_vs, generic_texgen_fs);
2116
2117 avnd::cpu_texture_output_introspection<T>::for_all(
2118 avnd::get_outputs<T>(*self.state), [&](auto& t) {
2119 self.m_samplers.push_back(
2120 createOutputTexture(renderer, t.texture, QSize{t.texture.width, t.texture.height}));
2121 });
2122
2123 gpu_first = self.m_samplers.size();
2124 avnd::gpu_texture_output_introspection<T>::for_all(
2125 avnd::get_outputs<T>(*self.state), [&](auto&) {
2126 auto sampler = renderer.state.rhi->newSampler(
2127 QRhiSampler::Linear, QRhiSampler::Linear, QRhiSampler::None,
2128 QRhiSampler::ClampToEdge, QRhiSampler::ClampToEdge);
2129 sampler->create();
2130 self.m_samplers.push_back(score::gfx::Sampler{sampler, nullptr});
2131 });
2132
2133 self.m_outputPremultiplied = drawnOutputPremultiplied(self);
2134 self.defaultPassesInit(renderer, mesh);
2135
2136 // defaultPassesInit only covers the edges of output[0]: an edge leaving any
2137 // other texture outlet (e.g. Image Processor's Mask / Depth) gets its pass
2138 // here, bound to that outlet's own sampler (see samplersForOutputEdge).
2139 for(auto* port : self.node().output)
2140 {
2141 if(!port || port->type != score::gfx::Types::Image)
2142 continue;
2143 for(auto* edge : port->edges)
2144 if(!self.hasOutputPassForEdge(*edge))
2145 self.addOutputPass(renderer, *edge, res);
2146 }
2147 }
2148
2149 void runInitialPasses(auto& self,
2150 score::gfx::RenderList& renderer,
2151 QRhiResourceUpdateBatch*& res)
2152 {
2153 avnd::cpu_texture_output_introspection<T>::for_all_n(
2154 avnd::get_outputs<T>(*self.state), [&]<std::size_t N>(auto& t, avnd::predicate_index<N>) {
2155 uploadOutputTexture(self, renderer, N, t.texture, res);
2156 });
2157
2158 std::size_t k = gpu_first;
2159 avnd::gpu_texture_output_introspection<T>::for_all(
2160 avnd::get_outputs<T>(*self.state), [&](auto& t) {
2161 auto* tex = static_cast<QRhiTexture*>(t.texture.handle);
2162 if(tex
2163 && (tex->flags()
2164 & (QRhiTexture::CubeMap | QRhiTexture::ThreeDimensional
2165 | QRhiTexture::TextureArray)))
2166 tex = nullptr;
2167 auto& sampler = self.m_samplers[k];
2168 if(tex != sampler.texture)
2169 {
2170 sampler.texture = tex;
2171 for(auto& [edge, pass] : self.m_p)
2172 {
2173 if(!pass.p.srb || !edge)
2174 continue;
2175 // A pass bound to its edge's own sampler (samplersForOutputEdge) holds
2176 // it at binding 3; a pass holding every sampler has this one at 3 + k.
2177 const auto bound = self.samplersForOutputEdge(*edge);
2178 if(bound.size() == 1)
2179 {
2180 if(bound.data() == &sampler)
2181 score::gfx::replaceTexture(
2182 *pass.p.srb, 3, tex ? tex : &renderer.emptyTexture());
2183 }
2184 else
2185 {
2186 score::gfx::replaceTexture(
2187 *pass.p.srb, int(3 + k), tex ? tex : &renderer.emptyTexture());
2188 }
2189 }
2190 }
2191 k++;
2192 });
2193 }
2194
2195 void release(auto& self, score::gfx::RenderList& r)
2196 {
2197 for(std::size_t i = 0; i < self.m_samplers.size(); i++)
2198 {
2199 auto& texture = self.m_samplers[i].texture;
2200 if(i < gpu_first && texture != &r.emptyTexture())
2201 texture->deleteLater();
2202 texture = nullptr;
2203 }
2204 }
2205
2206 std::size_t gpu_first{};
2207
2208};
2209
2210template<typename T>
2211 requires (avnd::texture_output_introspection<T>::size == 0)
2212struct texture_outputs_storage<T>
2213{
2214 static void init(auto& self, score::gfx::RenderList& renderer, QRhiResourceUpdateBatch& res)
2215 {
2216 }
2217
2218 static void runInitialPasses(auto& self,
2219 score::gfx::RenderList& renderer,
2220 QRhiResourceUpdateBatch*& res)
2221 {
2222 }
2223
2224 static void release(auto& self, score::gfx::RenderList& r)
2225 {
2226 }
2227};
2228template<typename T>
2229struct geometry_outputs_storage;
2230
2231template<typename T>
2232 requires (avnd::geometry_output_introspection<T>::size > 0)
2233struct geometry_outputs_storage<T>
2234{
2235 ossia::geometry_spec specs[avnd::geometry_output_introspection<T>::size];
2236
2237 struct sent_transform
2238 {
2239 std::optional<ossia::transform3d> value;
2240 std::vector<std::pair<const score::gfx::Edge*, const score::gfx::NodeRenderer*>>
2241 receivers;
2242 };
2243 sent_transform transforms[avnd::geometry_output_introspection<T>::size];
2244
2245 template <avnd::geometry_port Field>
2246 void reload_mesh(Field& ctrl, ossia::geometry_spec& spc)
2247 {
2248 spc.meshes = std::make_shared<ossia::mesh_list>();
2249 auto& ossia_meshes = *spc.meshes;
2250 if constexpr(avnd::static_geometry_type<Field> || avnd::dynamic_geometry_type<Field>)
2251 {
2252 ossia_meshes.meshes.resize(1);
2253 load_geometry(ctrl, ossia_meshes.meshes[0]);
2254 }
2255 else if constexpr(
2256 avnd::static_geometry_type<decltype(Field::mesh)>
2257 || avnd::dynamic_geometry_type<decltype(Field::mesh)>)
2258 {
2259 ossia_meshes.meshes.resize(1);
2260 load_geometry(ctrl.mesh, ossia_meshes.meshes[0]);
2261 }
2262 else
2263 {
2264 load_geometry(ctrl, ossia_meshes);
2265 }
2266 }
2267
2268 template <avnd::geometry_port Field, std::size_t N>
2269 void upload(
2270 score::gfx::RenderList& renderer, Field& ctrl, score::gfx::Edge& edge,
2271 avnd::predicate_index<N>)
2272 {
2273 auto edge_sink = edge.sink;
2274 if(auto pnode = edge_sink->node)
2275 {
2276 ossia::geometry_spec& spc = specs[N];
2277
2278 // 1. Reload mesh
2279 {
2280 if(ctrl.dirty_mesh)
2281 {
2282 reload_mesh(ctrl, spc);
2283 }
2284 else
2285 {
2286 if(spc.meshes)
2287 {
2288 auto& ossia_meshes = *spc.meshes;
2289
2290 bool any_need_reload = false;
2291 bool any_need_upload = false;
2292 if constexpr(avnd::static_geometry_type<Field> || avnd::dynamic_geometry_type<Field>)
2293 {
2294 SCORE_ASSERT(ossia_meshes.meshes.size() == 1);
2295 auto [need_reload, need_upload]
2296 = update_geometry(ctrl, ossia_meshes.meshes[0]);
2297 any_need_reload = need_reload;
2298 any_need_upload = need_upload;
2299 }
2300 else if constexpr(
2301 avnd::static_geometry_type<decltype(Field::mesh)>
2302 || avnd::dynamic_geometry_type<decltype(Field::mesh)>)
2303 {
2304 SCORE_ASSERT(ossia_meshes.meshes.size() == 1);
2305 auto [need_reload, need_upload]
2306 = update_geometry(ctrl.mesh, ossia_meshes.meshes[0]);
2307 any_need_reload = need_reload;
2308 any_need_upload = need_upload;
2309 }
2310 else
2311 {
2312 auto [need_reload, need_upload] = update_geometry(ctrl, ossia_meshes);
2313 any_need_reload = need_reload;
2314 any_need_upload = need_upload;
2315 }
2316
2317 if(any_need_reload)
2318 {
2319 reload_mesh(ctrl, spc);
2320 }
2321 }
2322 }
2323 ctrl.dirty_mesh = false;
2324 }
2325
2326 // 2. Push to next node
2327 // FIXME this should be for the renderer of edge, not the node, since
2328 // geometries can have gpu buffers
2329 auto rendered_node = pnode->renderedNodes.find(&renderer);
2330 SCORE_ASSERT(rendered_node != pnode->renderedNodes.end());
2331
2332 auto it = std::find(
2333 edge_sink->node->input.begin(), edge_sink->node->input.end(), edge_sink);
2334 SCORE_ASSERT(it != edge_sink->node->input.end());
2335 int n = it - edge_sink->node->input.begin();
2336
2337 rendered_node->second->process(n, spc, edge.source);
2338
2339 // 3. Same for transform3d
2340
2341 if constexpr(requires { ctrl.transform; })
2342 {
2343 auto& sent = transforms[N];
2344 if(ctrl.dirty_transform)
2345 {
2346 sent.value.emplace();
2347 std::copy_n(ctrl.transform, std::ssize(ctrl.transform), sent.value->matrix);
2348 sent.receivers.clear();
2349 ctrl.dirty_transform = false;
2350 }
2351
2352 const std::pair<const score::gfx::Edge*, const score::gfx::NodeRenderer*>
2353 receiver{&edge, rendered_node->second};
2354 if(sent.value && !ossia::contains(sent.receivers, receiver))
2355 {
2356 sent.receivers.push_back(receiver);
2357 rendered_node->second->process(n, *sent.value);
2358 if(auto pnode = dynamic_cast<score::gfx::ProcessNode*>(edge_sink->node))
2359 pnode->process(n, *sent.value);
2360 }
2361 }
2362 }
2363 }
2364
2365 void upload(score::gfx::RenderList& renderer, auto& state, score::gfx::Edge& edge)
2366 {
2367 // FIXME we need something such as port_run_{pre,post}process for GPU nodes
2368 avnd::geometry_output_introspection<T>::for_all_n(
2369 avnd::get_outputs(state),
2370 [&](auto& field, auto pred) { this->upload(renderer, field, edge, pred); });
2371 }
2372
2373 // Lifecycle parity with the other *_outs storages. The geometry_spec
2374 // wrapper carries non-owning pointers + transform values, so release has
2375 // nothing to do; it exists so RHI handles added later have a hook.
2376 void release(score::gfx::RenderList&) noexcept { }
2377};
2378
2379
2380template<typename T>
2381 requires (avnd::geometry_output_introspection<T>::size == 0)
2382struct geometry_outputs_storage<T>
2383{
2384 static void upload(auto&&...)
2385 {
2386
2387 }
2388 static void release(auto&&...) noexcept { }
2389};
2390
2391// Scene output support (Crousti-side pending promotion to avendish).
2392// The `scene_port` concept and `scene_dirt_flags` live in SceneConcepts.hpp
2393// so the port-creation visitor in ProcessModelPortInit.hpp can reuse them.
2394
2395template <typename Field>
2396using is_scene_port_t = boost::mp11::mp_bool<scene_port<Field>>;
2397
2398template <typename T>
2399using scene_output_introspection =
2400 avnd::predicate_introspection<typename avnd::outputs_type<T>::type, is_scene_port_t>;
2401
2402template <typename T>
2403using scene_input_introspection =
2404 avnd::predicate_introspection<typename avnd::inputs_type<T>::type, is_scene_port_t>;
2405
2406// Scene input transport: NodeRenderer::process(port, scene_spec, source)
2407// keeps one scene per (port, source) and merges them all into `this->scene`.
2408// A node with a single scene input gets that merged scene. With several, each
2409// field receives only the scenes that arrived on its own port, merged across
2410// sources; the merge is memoized per port on the (state, version) set so an
2411// unchanged input keeps its scene_state identity.
2412template <typename T>
2413struct scene_inputs_storage;
2414
2415template <typename T>
2416 requires(scene_input_introspection<T>::size > 0)
2417struct scene_inputs_storage<T>
2418{
2419 struct port_cache
2420 {
2421 ossia::small_vector<std::pair<const ossia::scene_state*, int64_t>, 4> inputs;
2422 ossia::scene_spec merged;
2423 };
2424 port_cache m_cache[scene_input_introspection<T>::size];
2425
2426 static ossia::scene_spec
2427 sceneOnPort(const score::gfx::NodeRenderer& renderer, int port, port_cache& cache)
2428 {
2429 ossia::small_vector<std::pair<const ossia::scene_state*, int64_t>, 4> sig;
2430 ossia::small_vector<ossia::scene_spec, 4> scenes;
2431 renderer.forEachSceneOnPort(port, [&](const ossia::scene_spec& s) {
2432 sig.push_back({s.state.get(), s.state->version});
2433 scenes.push_back(s);
2434 });
2435
2436 if(scenes.empty())
2437 {
2438 cache = {};
2439 return {};
2440 }
2441 if(scenes.size() == 1)
2442 {
2443 cache = {};
2444 return scenes[0];
2445 }
2446 if(sig == cache.inputs && cache.merged.state)
2447 return cache.merged;
2448
2449 cache.inputs.assign(sig.begin(), sig.end());
2450 cache.merged = ossia::merge_scenes(
2451 std::span<const ossia::scene_spec>{scenes.data(), scenes.size()});
2452 return cache.merged;
2453 }
2454
2455 void readInputScenes(const score::gfx::NodeRenderer& renderer, auto& state)
2456 {
2457 if constexpr(scene_input_introspection<T>::size == 1)
2458 {
2459 scene_input_introspection<T>::for_all(
2460 avnd::get_inputs<T>(state), [&](auto& field) { field.scene = renderer.scene; });
2461 return;
2462 }
2463 scene_input_introspection<T>::for_all_n2(
2464 avnd::get_inputs<T>(state),
2465 [&]<typename F, std::size_t K, std::size_t N>(
2466 F& field, avnd::predicate_index<K>, avnd::field_index<N>) {
2467 field.scene = sceneOnPort(renderer, int(N), m_cache[K]);
2468 });
2469 }
2470
2471 void release(score::gfx::RenderList&)
2472 {
2473 for(auto& c : m_cache)
2474 c = {};
2475 }
2476};
2477
2478template <typename T>
2479 requires(scene_input_introspection<T>::size == 0)
2480struct scene_inputs_storage<T>
2481{
2482 static void readInputScenes(auto&&...) { }
2483 static void release(auto&&...) { }
2484};
2485
2486template <typename T>
2487struct scene_outputs_storage;
2488
2489template <typename T>
2490 requires(scene_output_introspection<T>::size > 0)
2491struct scene_outputs_storage<T>
2492{
2493 template <scene_port Field, std::size_t N>
2494 void upload(
2495 score::gfx::RenderList& renderer, Field& ctrl, score::gfx::Edge& edge,
2496 avnd::predicate_index<N>)
2497 {
2498 // Publish the scene every frame rather than only when `ctrl.dirty` is
2499 // set: a once-only push gets overwritten by any other producer on the
2500 // same downstream inlet. Consumers short-circuit on shared_ptr
2501 // identity + version, so publishing every frame costs only refcount
2502 // bumps.
2503 if(!ctrl.scene.state)
2504 return;
2505
2506 auto* edge_sink = edge.sink;
2507 if(!edge_sink || !edge_sink->node)
2508 return;
2509
2510 auto rendered_node = edge_sink->node->renderedNodes.find(&renderer);
2511 if(rendered_node == edge_sink->node->renderedNodes.end())
2512 return;
2513
2514 auto it = std::find(
2515 edge_sink->node->input.begin(), edge_sink->node->input.end(), edge_sink);
2516 if(it == edge_sink->node->input.end())
2517 return;
2518 int n = it - edge_sink->node->input.begin();
2519
2520 // NodeRenderer::process(port, scene_spec, source_key) handles additive
2521 // merging across multiple producers converging on the same sink port
2522 // (keyed on the source edge's producer Port pointer), extracts a legacy
2523 // geometry_spec for downstream consumers that only understand geometry,
2524 // and sets sceneChanged=true.
2525 rendered_node->second->process(n, ctrl.scene, edge.source);
2526
2527 if constexpr(requires { ctrl.dirty; })
2528 ctrl.dirty = 0;
2529 }
2530
2531 void upload(score::gfx::RenderList& renderer, auto& state, score::gfx::Edge& edge)
2532 {
2533 scene_output_introspection<T>::for_all_n(
2534 avnd::get_outputs(state),
2535 [&](auto& field, auto pred) { this->upload(renderer, field, edge, pred); });
2536 }
2537
2538 // Lifecycle parity with texture_outputs_storage / buffer_outputs_storage:
2539 // the storage owns no QRhi resources (the scene_spec is a value-semantics
2540 // struct plus a shared_ptr to scene_state, both managed by their own
2541 // destructors), so release has nothing to do. It keeps CpuFilterNode /
2542 // CpuAnalysisNode releaseState symmetric across all storages, and gives
2543 // RHI handles added later a hook.
2544 void release(score::gfx::RenderList&) noexcept { }
2545};
2546
2547template <typename T>
2548 requires(scene_output_introspection<T>::size == 0)
2549struct scene_outputs_storage<T>
2550{
2551 static void upload(auto&&...) { }
2552 static void release(auto&&...) noexcept { }
2553};
2554
2555}
2556
2557#endif
Root data model for visual nodes.
Definition score-plugin-gfx/Gfx/Graph/Node.hpp:76
std::vector< Port * > input
Input ports of that node.
Definition score-plugin-gfx/Gfx/Graph/Node.hpp:105
bool hasExplicitRenderTargetSpecs(int32_t port) const noexcept
Whether the user set a size or a format on a texture inlet.
Definition score-plugin-gfx/Gfx/Graph/Node.hpp:184
ossia::flat_map< RenderList *, score::gfx::NodeRenderer * > renderedNodes
Map associating each RenderList to a Renderer for this model.
Definition score-plugin-gfx/Gfx/Graph/Node.hpp:116
virtual void process(Message &&msg)
Process a message from the execution engine.
Definition Node.cpp:25
ossia::small_pod_vector< Port *, 1 > output
Output ports of that node.
Definition score-plugin-gfx/Gfx/Graph/Node.hpp:111
Common base class for most single-pass, simple nodes.
Definition score-plugin-gfx/Gfx/Graph/Node.hpp:279
Renderer for a given node.
Definition NodeRenderer.hpp:11
ossia::scene_spec scene
The scene to use (when receiving scene_spec data).
Definition NodeRenderer.hpp:215
void forEachSceneOnPort(int32_t port, F &&fn) const
Definition NodeRenderer.hpp:172
Base class for sink nodes (QWindow, spout, syphon, NDI output, ...)
Definition OutputNode.hpp:55
Common base class for nodes that map to score processes.
Definition score-plugin-gfx/Gfx/Graph/Node.hpp:216
List of nodes to be rendered to an output.
Definition RenderList.hpp:33
RenderTargetSpecs resolveInputRenderTargetSpecs(const Node &node, int32_t port) noexcept
The render target specs of a texture inlet, with its size resolved.
Definition RenderList.cpp:366
bool requiresDepth(const score::gfx::Port &p) const noexcept
Whether this list of rendering actions requires depth testing at all.
Definition RenderList.cpp:1709
const score::gfx::Mesh & defaultTriangle() const noexcept
A triangle mesh correct for this API.
Definition RenderList.cpp:1748
RenderState & state
RenderState corresponding to this RenderList.
Definition RenderList.hpp:274
QRhiTexture & emptyTexture() const noexcept
Texture to use when a texture is missing (2D)
Definition RenderList.hpp:297
void releaseBuffer(QRhiBuffer *buf)
Retire a buffer this render list's node owns.
Definition RenderList.cpp:1084
TreeNode< DeviceExplorerNode > Node
Definition DeviceNode.hpp:74
void settle(::State::Address &a, const score::DocumentContext &ctx)
Definition ScriptableReference.cpp:121
Definition Controls.hpp:27
void bind_worker(Object &obj, std::weak_ptr< Object > obj_wp, Delivery deliver)
Definition WorkerBinding.hpp:31
std::pair< QShader, QShader > makeShaders(const RenderState &v, QString vert, QString frag, int multiViewCount)
Get a pair of compiled vertex / fragment shaders from GLSL 4.5 sources.
Definition score-plugin-gfx/Gfx/Graph/Utils.cpp:1359
TextureRenderTarget createRenderTarget(const RenderState &state, QRhiTexture *tex, int samples, bool depth, bool samplableDepth)
Create a render target from a texture.
Definition score-plugin-gfx/Gfx/Graph/Utils.cpp:68
void uploadStaticBufferWithStoredData(QRhiResourceUpdateBatch *ub, QRhiBuffer *buf, int offset, int64_t bytesize, const char *data)
Schedule a Static buffer update when we can guarantee the buffer outlives the frame.
Definition score-plugin-gfx/Gfx/Graph/Utils.hpp:933
QRhiBuffer::UsageFlags compatibleBufferUsage(QRhi &rhi, QRhiBuffer::UsageFlags usage) noexcept
Drop a StorageBuffer usage the backend cannot actually honour.
Definition RenderState.hpp:223
Base toolkit upon which the software is built.
Definition Application.cpp:108
STL namespace.
Definition DocumentContext.hpp:18
Definition Mesh.hpp:18
Connection between two score::gfx::Port.
Definition score-plugin-gfx/Gfx/Graph/Utils.hpp:220
Definition score-plugin-gfx/Gfx/Graph/Node.hpp:51
Definition OutputNode.hpp:16
Port of a score::gfx::Node.
Definition score-plugin-gfx/Gfx/Graph/Utils.hpp:199
score::gfx::Node * node
Parent node of the port.
Definition score-plugin-gfx/Gfx/Graph/Utils.hpp:201
Definition score-plugin-gfx/Gfx/Graph/Node.hpp:58
Stores a sampler and the texture currently associated with it.
Definition score-plugin-gfx/Gfx/Graph/Utils.hpp:164
Useful abstraction for storing all the data related to a render target.
Definition score-plugin-gfx/Gfx/Graph/Utils.hpp:279