checksums.yaml ADDED
@@ -0,0 +1,7 @@
1
+ ---
2
+ SHA256:
3
+ metadata.gz: 28fd9758cdc95babbcf72463572301d88e9691de7f90c927bd53db60dc2b46b4
4
+ data.tar.gz: a70e2704a22a10f1f2a68843794099dfaeb88d1ad08e0734f39ec1711f5975a3
5
+ SHA512:
6
+ metadata.gz: fcff6081b20d4d551a74eb094b1b6e5e61149a31bf367dbf693c88c12102b0c9fe032fe7c0766e09b1d5b63fc44393de128cc94c7d9a01e6c867814938fdf072
7
+ data.tar.gz: 10c62b40edbad82555426ab3ba2bd514ff7904e5e04c775aa22b5b2f63c0034ba2360a67f26a3e5aa53e160625834cc39e92a0ad78602af8e5ece025bbb0f0f9
data/CHANGELOG.md ADDED
@@ -0,0 +1,9 @@
1
+ # Data Forest Changelog
2
+
3
+ ## 0.0.2
4
+
5
+ No need to explicitly define leaf nodes if they are trivial.
6
+
7
+ ## 0.0.1
8
+
9
+ Initial version. Allows you to define and execute basic data processing trees.
data/Gemfile ADDED
@@ -0,0 +1,6 @@
1
+ # frozen_string_literal: true
2
+
3
+ source 'https://rubygems.org'
4
+
5
+ # Specify your gem's dependencies in data-forest.gemspec
6
+ gemspec
data/LICENSE ADDED
@@ -0,0 +1,29 @@
1
+ BSD 3-Clause License
2
+
3
+ Copyright (c) 2019, Codruț Constantin Gușoi
4
+ All rights reserved.
5
+
6
+ Redistribution and use in source and binary forms, with or without
7
+ modification, are permitted provided that the following conditions are met:
8
+
9
+ * Redistributions of source code must retain the above copyright notice, this
10
+ list of conditions and the following disclaimer.
11
+
12
+ * Redistributions in binary form must reproduce the above copyright notice,
13
+ this list of conditions and the following disclaimer in the documentation
14
+ and/or other materials provided with the distribution.
15
+
16
+ * Neither the name of the copyright holder nor the names of its
17
+ contributors may be used to endorse or promote products derived from
18
+ this software without specific prior written permission.
19
+
20
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
21
+ AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
22
+ IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
23
+ DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
24
+ FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
25
+ DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
26
+ SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
27
+ CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
28
+ OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
29
+ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
data/README.md ADDED
@@ -0,0 +1,132 @@
1
+ # Data Forest
2
+
3
+ Execute data processing jobs, defined as a dependency graph.
4
+
5
+ ## Usage
6
+
7
+ Define a data forest with a tree structure:
8
+
9
+ ```ruby
10
+ DataForest.define(:my_forest) do
11
+ tree :my_tree do
12
+ root root_node: [:node_one, :node_two]
13
+
14
+ node node_one: :node_three
15
+ end
16
+ end
17
+ ```
18
+
19
+ Define your jobs:
20
+
21
+ ```ruby
22
+ class RootNode < DataForest::Job
23
+ def perform
24
+ puts self.class.name
25
+ puts input(:node_one) + input(:node_two)
26
+ end
27
+ end
28
+ class NodeOne < DataForest::Job
29
+ def perform
30
+ puts self.class.name
31
+ output(1 + input(:node_three))
32
+ end
33
+ end
34
+ class NodeTwo < DataForest::Job
35
+ def perform
36
+ puts self.class.name
37
+ output(2)
38
+ end
39
+ end
40
+ class NodeThree < DataForest::Job
41
+ def perform
42
+ puts self.class.name
43
+ output(3)
44
+ end
45
+ end
46
+ ```
47
+
48
+ When you execute:
49
+
50
+ ```ruby
51
+ DataForest.run(:my_forest, :my_tree)
52
+ ```
53
+
54
+ You get the following output:
55
+
56
+ ```
57
+ NodeThree
58
+ NodeTwo
59
+ NodeOne
60
+ RootNode
61
+ 6
62
+ ```
63
+
64
+ The outputs are cached in between runs, so if executed again, you get:
65
+
66
+ ```
67
+ RootNode
68
+ 6
69
+ ```
70
+
71
+ You can use the CLI to run examples:
72
+
73
+ ```sh
74
+ data-forest -f ./examples/simple_tree.rb run my_forest:simple_tree
75
+ ```
76
+
77
+ ## Install
78
+
79
+ Add this line to your application's Gemfile:
80
+
81
+ ```ruby
82
+ gem 'data-forest'
83
+ ```
84
+
85
+ And then execute:
86
+
87
+ ```sh
88
+ bundle install
89
+ ```
90
+
91
+ Or install it yourself as:
92
+
93
+ ```sh
94
+ gem install data-forest
95
+ ```
96
+
97
+ ## Development
98
+
99
+ ### Lint
100
+
101
+ ```sh
102
+ bundle exec rubocop
103
+ ```
104
+
105
+ ### Test
106
+
107
+ ```sh
108
+ bundle exec rake test
109
+ ```
110
+
111
+ ### Examples
112
+
113
+ ```sh
114
+ bundle exec data-forest -f ./examples/simple_tree.rb run my_forest:simple_tree
115
+ bundle exec data-forest -f ./examples/valid_tree.rb run my_forest:valid_tree
116
+ bundle exec data-forest -f ./examples/mini_tree.rb run my_forest:mini_tree
117
+ ```
118
+
119
+ ## Publishing
120
+
121
+ ```
122
+ gem build data-forest
123
+ ls | grep data-forest- | xargs envchain rubygems gem push
124
+ ```
125
+
126
+ ## Contributing
127
+
128
+ Pull requests are welcome on GitLab at https://gitlab.com/sdwolfz/data-forest
129
+
130
+ ## License
131
+
132
+ The gem is available as open source under the terms of the [3-Clause BSD License](https://opensource.org/licenses/BSD-3-Clause).
data/bin/data-forest ADDED
@@ -0,0 +1,7 @@
1
+ #!/usr/bin/env ruby
2
+ # frozen_string_literal: true
3
+
4
+ require 'data-forest'
5
+ require 'data-forest/cli'
6
+
7
+ DataForest::CLI.run(ARGV)
data/data-forest.gemspec ADDED
@@ -0,0 +1,65 @@
1
+ # frozen_string_literal: true
2
+
3
+ lib = File.expand_path('lib', __dir__)
4
+ $LOAD_PATH.unshift(lib) unless $LOAD_PATH.include?(lib)
5
+ require 'data-forest'
6
+
7
+ Gem::Specification.new do |spec|
8
+ spec.name = 'data-forest'
9
+ spec.version = DataForest::VERSION
10
+ spec.authors = ['Codruț Constantin Gușoi']
11
+ spec.email = ['codrut.gusoi@gmail.com']
12
+ spec.summary = 'Define and execute data processing trees.'
13
+ spec.homepage = 'https://gitlab.com/sdwolfz/data-forest'
14
+ spec.license = 'BSD-3-Clause'
15
+
16
+ spec.metadata['homepage_uri'] = spec.homepage
17
+ spec.metadata['source_code_uri'] = 'https://gitlab.com/sdwolfz/data-forest'
18
+ spec.metadata['changelog_uri'] = 'https://gitlab.com/sdwolfz/data-forest/blob/master/CHANGELOG.md'
19
+
20
+ spec.files = [
21
+ # Documentation
22
+ 'CHANGELOG.md',
23
+ 'README.md',
24
+
25
+ # Legal
26
+ 'LICENSE',
27
+
28
+ # Gem
29
+ 'Gemfile',
30
+ 'data-forest.gemspec',
31
+
32
+ # CLI
33
+ 'bin/data-forest',
34
+ 'lib/data-forest/cli.rb',
35
+ 'lib/data-forest/cli/command.rb',
36
+ 'lib/data-forest/cli/command/main.rb',
37
+ 'lib/data-forest/cli/command/run.rb',
38
+ 'lib/data-forest/cli/runner.rb',
39
+
40
+ # Core
41
+ 'lib/data-forest.rb',
42
+ 'lib/data-forest/builder.rb',
43
+ 'lib/data-forest/cache/base.rb',
44
+ 'lib/data-forest/cache/filesystem.rb',
45
+ 'lib/data-forest/cache/memory.rb',
46
+ 'lib/data-forest/forest.rb',
47
+ 'lib/data-forest/job.rb',
48
+ 'lib/data-forest/node.rb',
49
+ 'lib/data-forest/runner.rb',
50
+ 'lib/data-forest/tree.rb'
51
+ ]
52
+
53
+ spec.bindir = 'bin'
54
+ spec.executables = spec.files.grep(%r{^bin/}) { |f| File.basename(f) }
55
+ spec.require_paths = ['lib']
56
+
57
+ spec.add_development_dependency 'bundler', '~> 2.0'
58
+ spec.add_development_dependency 'irb', '~> 1.1.0'
59
+ spec.add_development_dependency 'minitest', '~> 5.0'
60
+ spec.add_development_dependency 'pry-byebug', '~> 3.7.0'
61
+ spec.add_development_dependency 'pry-doc', '~> 1.0.0'
62
+ spec.add_development_dependency 'rake', '~> 13.0.1'
63
+ spec.add_development_dependency 'rubocop', '~> 0.77.0'
64
+ spec.add_development_dependency 'solargraph', '~> 0.38.0'
65
+ end
data/lib/data-forest.rb ADDED
@@ -0,0 +1,23 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'data-forest/builder'
4
+ require 'data-forest/job'
5
+ require 'data-forest/runner'
6
+
7
+ module DataForest
8
+ VERSION = '0.0.2'
9
+
10
+ FORESTS = {}
11
+
12
+ def self.define(name, &block)
13
+ FORESTS[name] = Builder.new(name, block).build
14
+
15
+ self
16
+ end
17
+
18
+ def self.run(name, tree_name)
19
+ Runner.new.call(FORESTS[name], tree_name)
20
+
21
+ self
22
+ end
23
+ end
data/lib/data-forest/builder.rb ADDED
@@ -0,0 +1,124 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'data-forest/forest'
4
+ require 'data-forest/node'
5
+ require 'data-forest/tree'
6
+
7
+ module DataForest
8
+ class Builder
9
+ def initialize(name, block)
10
+ @name = name
11
+ @block = block
12
+
13
+ @forest = nil
14
+ @nodes = nil
15
+ @root = nil
16
+ @trees = nil
17
+ end
18
+
19
+ def build
20
+ @trees = {}
21
+
22
+ instance_exec(&@block)
23
+ @forest = Forest.new(@name, @trees)
24
+ @trees = nil
25
+
26
+ @forest
27
+ end
28
+
29
+ def tree(name, &block)
30
+ @nodes = {}
31
+ @root = nil
32
+
33
+ instance_exec(&block)
34
+
35
+ remaining = []
36
+ @nodes.each do |_, node|
37
+ node.dependencies.each do |dependency|
38
+ remaining.push(dependency)
39
+ end
40
+ end
41
+ remaining -= @nodes.keys
42
+ remaining.each { |node| add_node_to_list(node) }
43
+
44
+ @trees[name] = Tree.new(name, @root, @nodes)
45
+
46
+ @nodes = nil
47
+ @root = nil
48
+
49
+ self
50
+ end
51
+
52
+ def root(args)
53
+ @root = add_node_to_list(args)
54
+
55
+ self
56
+ end
57
+
58
+ def node(args)
59
+ add_node_to_list(args)
60
+
61
+ self
62
+ end
63
+
64
+ private
65
+
66
+ def extract_name(args)
67
+ name = nil
68
+
69
+ case args
70
+ when Symbol
71
+ name = args
72
+ when Hash
73
+ keys = args.keys
74
+
75
+ raise ArgumentError if keys.size != 1
76
+
77
+ name = keys.first
78
+ else
79
+ raise ArgumentError
80
+ end
81
+
82
+ name
83
+ end
84
+
85
+ def extract_dependencies(name, args)
86
+ dependencies = nil
87
+
88
+ case args
89
+ when Symbol
90
+ dependencies = []
91
+ when Hash
92
+ dependencies = case args[name]
93
+ when Symbol
94
+ [args[name]]
95
+ when Array
96
+ args[name]
97
+ else
98
+ raise ArgumentError
99
+ end
100
+ else
101
+ raise ArgumentError
102
+ end
103
+
104
+ dependencies
105
+ end
106
+
107
+ def name_to_job_class(name)
108
+ result = Object.const_get(name.to_s.split('_').map(&:capitalize).join)
109
+ raise ArgumentError unless result.ancestors.include?(DataForest::Job)
110
+
111
+ result
112
+ end
113
+
114
+ def add_node_to_list(args)
115
+ name = extract_name(args)
116
+ dependencies = extract_dependencies(name, args)
117
+ job_class = name_to_job_class(name)
118
+
119
+ @nodes[name] = Node.new(name, job_class, dependencies)
120
+
121
+ name
122
+ end
123
+ end
124
+ end
data/lib/data-forest/cache/base.rb ADDED
@@ -0,0 +1,32 @@
1
+ # frozen_string_literal: true
2
+
3
+ module DataForest
4
+ module Cache
5
+ class Base
6
+ def initialize(forest_name, tree_name)
7
+ @forest_name = forest_name
8
+ @tree_name = tree_name
9
+ end
10
+
11
+ def expire!
12
+ raise NotImplementedError
13
+ end
14
+
15
+ def find!(_key)
16
+ raise NotImplementedError
17
+ end
18
+
19
+ def key?(_key)
20
+ raise NotImplementedError
21
+ end
22
+
23
+ def setup!
24
+ raise NotImplementedError
25
+ end
26
+
27
+ def store!(_key, _value)
28
+ raise NotImplementedError
29
+ end
30
+ end
31
+ end
32
+ end
data/lib/data-forest/cache/filesystem.rb ADDED
@@ -0,0 +1,57 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'data-forest/cache/base'
4
+
5
+ require 'fileutils'
6
+
7
+ module DataForest
8
+ module Cache
9
+ class Filesystem < Base
10
+ def initialize(forest_name, tree_name)
11
+ super
12
+
13
+ @cache_directory_name = 'cache'
14
+ @cache_directory_root = Dir.pwd
15
+ end
16
+
17
+ def expire!
18
+ FileUtils.rm_rf(forest_directory_path)
19
+ end
20
+
21
+ def find!(key)
22
+ File.read(cache_file_path(key))
23
+ end
24
+
25
+ def key?(key)
26
+ File.exist?(cache_file_path(key))
27
+ end
28
+
29
+ def setup!
30
+ create_cache_directories!
31
+ end
32
+
33
+ def store!(key, value)
34
+ File.write(cache_file_path(key), value)
35
+ end
36
+
37
+ private
38
+
39
+ def create_cache_directories!
40
+ FileUtils.mkdir_p(cache_directory_path)
41
+ end
42
+
43
+ def cache_directory_path
44
+ File.join(forest_directory_path, @tree_name.to_s)
45
+ end
46
+
47
+ def forest_directory_path
48
+ File.join(@cache_directory_root, @cache_directory_name, @forest_name.to_s)
49
+ end
50
+
51
+ def cache_file_path(key)
52
+ file_name = key.to_s + '.json'
53
+ File.join(cache_directory_path, file_name)
54
+ end
55
+ end
56
+ end
57
+ end
data/lib/data-forest/cache/memory.rb ADDED
@@ -0,0 +1,42 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'data-forest/cache/base'
4
+
5
+ module DataForest
6
+ module Cache
7
+ class Memory < Base
8
+ def initialize(forest_name, tree_name)
9
+ super
10
+
11
+ @namespace = [forest_name, tree_name].map(&:to_s).join('-')
12
+ @cache = {}
13
+ end
14
+
15
+ def expire!
16
+ @cache = {}
17
+ end
18
+
19
+ def find!(key)
20
+ @cache[cache_key(key)]
21
+ end
22
+
23
+ def key?(key)
24
+ @cache.key?(cache_key(key))
25
+ end
26
+
27
+ def setup!
28
+ # Nothing to do
29
+ end
30
+
31
+ def store!(key, value)
32
+ @cache[cache_key(key)] = value
33
+ end
34
+
35
+ private
36
+
37
+ def cache_key(key)
38
+ [@namespace, key.to_s].join('-')
39
+ end
40
+ end
41
+ end
42
+ end
data/lib/data-forest/cli.rb ADDED
@@ -0,0 +1,11 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'data-forest/cli/runner'
4
+
5
+ module DataForest
6
+ module CLI
7
+ def self.run(argv)
8
+ Runner.new(argv).call
9
+ end
10
+ end
11
+ end
data/lib/data-forest/cli/command.rb ADDED
@@ -0,0 +1,11 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'data-forest/cli/command/main'
4
+ require 'data-forest/cli/command/run'
5
+
6
+ module DataForest
7
+ module CLI
8
+ module Command
9
+ end
10
+ end
11
+ end
data/lib/data-forest/cli/command/main.rb ADDED
@@ -0,0 +1,46 @@
1
+ # frozen_string_literal: true
2
+
3
+ module DataForest
4
+ module CLI
5
+ module Command
6
+ class Main
7
+ def options!(argv)
8
+ options = {}
9
+
10
+ parser = OptionParser.new do |opts|
11
+ opts.banner = 'Usag: data-forest [options]'
12
+
13
+ short_flag = '-f PATH'
14
+ long_flag = '--file PATH'
15
+ type = String
16
+ description = 'Path to DataForest definition file'
17
+
18
+ opts.on(short_flag, long_flag, type, description) do |path|
19
+ options[:path] = path
20
+ end
21
+
22
+ opts.on('-v', '--version', 'Display version') do |version|
23
+ options[:version] = version
24
+ end
25
+ end
26
+
27
+ parser.order!(argv)
28
+
29
+ options
30
+ end
31
+
32
+ def act!(options, _argv)
33
+ acted = false
34
+
35
+ if options[:version]
36
+ puts DataForest::VERSION
37
+
38
+ acted = true
39
+ end
40
+
41
+ acted
42
+ end
43
+ end
44
+ end
45
+ end
46
+ end
data/lib/data-forest/cli/command/run.rb ADDED
@@ -0,0 +1,50 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'pathname'
4
+
5
+ module DataForest
6
+ module CLI
7
+ module Command
8
+ class Run
9
+ def options!(argv)
10
+ options = {}
11
+
12
+ parser = OptionParser.new do |opts|
13
+ opts.banner = 'Usage: data-forest run FOREST_NAME:TREE_NAME'
14
+ end
15
+
16
+ parser.order!(argv)
17
+
18
+ options
19
+ end
20
+
21
+ def act!(options, argv)
22
+ id = argv.shift
23
+
24
+ names = id.split(':')
25
+ raise ArgumentError if names.size != 2
26
+
27
+ forest_name = names[0].to_sym
28
+ tree_name = names[1].to_sym
29
+
30
+ require_definition_file(options[:path])
31
+ DataForest.run(forest_name, tree_name)
32
+
33
+ true
34
+ end
35
+
36
+ private
37
+
38
+ def require_definition_file(path)
39
+ pathname = Pathname.new(path)
40
+
41
+ if pathname.absolute?
42
+ require 'path'
43
+ else
44
+ require File.join(Dir.pwd, pathname)
45
+ end
46
+ end
47
+ end
48
+ end
49
+ end
50
+ end
data/lib/data-forest/cli/runner.rb ADDED
@@ -0,0 +1,43 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'optparse'
4
+
5
+ require 'data-forest/cli/command'
6
+
7
+ module DataForest
8
+ module CLI
9
+ SUBCOMMANDS = {
10
+ 'run' => Command::Run
11
+ }.freeze
12
+
13
+ class Runner
14
+ def initialize(argv)
15
+ @argv = argv
16
+ @options = {}
17
+ end
18
+
19
+ # Returns `true` if a successful action was executed, `false` otherwise.
20
+ def call
21
+ acted = act_on_command(Command::Main)
22
+
23
+ return true if acted
24
+ return false if @argv.empty?
25
+
26
+ act_on_command(SUBCOMMANDS[@argv.shift])
27
+ end
28
+
29
+ private
30
+
31
+ def act_on_command(command)
32
+ return false unless command
33
+
34
+ command = command.new
35
+ options = command.options!(@argv)
36
+
37
+ @options.merge!(options)
38
+
39
+ command.act!(@options, @argv)
40
+ end
41
+ end
42
+ end
43
+ end
data/lib/data-forest/forest.rb ADDED
@@ -0,0 +1,40 @@
1
+ # frozen_string_literal: true
2
+
3
+ module DataForest
4
+ class Forest
5
+ attr_reader :name, :trees
6
+
7
+ def initialize(name, trees)
8
+ validate_name!(name)
9
+ validate_trees!(trees)
10
+
11
+ @name = name
12
+ @trees = trees
13
+ end
14
+
15
+ private
16
+
17
+ def validate_name!(name)
18
+ error = ArgumentError
19
+ message = 'name is not a Symbol'
20
+
21
+ raise(error, message) unless name.is_a?(Symbol)
22
+ end
23
+
24
+ def validate_trees!(trees)
25
+ error = ArgumentError
26
+ message = 'trees is not a Hash'
27
+
28
+ raise(error, message) unless trees.is_a?(Hash)
29
+
30
+ trees.each do |key, value|
31
+ message = "tree key: #{key} is not a Symbol, but a #{key.class.name}"
32
+ raise(error, message) unless key.is_a?(Symbol)
33
+
34
+ message = "tree value: #{value} is not a DataForest::Tree, "\
35
+ "but a #{value.class.name}"
36
+ raise(error, message) unless value.is_a?(DataForest::Tree)
37
+ end
38
+ end
39
+ end
40
+ end
data/lib/data-forest/job.rb ADDED
@@ -0,0 +1,47 @@
1
+ # frozen_string_literal: true
2
+
3
+ module DataForest
4
+ class Job
5
+ def initialize(forest_name, tree_name)
6
+ @forest_name = forest_name
7
+ @tree_name = tree_name
8
+
9
+ @has_input = false
10
+ @input = {}
11
+ @has_output = false
12
+ @output = nil
13
+ end
14
+
15
+ def add_input(key, value)
16
+ @has_input = true
17
+ @input[key] = value
18
+ end
19
+
20
+ def output?
21
+ @has_output
22
+ end
23
+
24
+ def retrieve_output
25
+ raise ArgumentError unless @has_output
26
+
27
+ @output
28
+ end
29
+
30
+ def perform
31
+ raise NotImplementedError
32
+ end
33
+
34
+ private
35
+
36
+ def input(key)
37
+ raise ArgumentError unless @has_input
38
+
39
+ @input[key]
40
+ end
41
+
42
+ def output(value)
43
+ @has_output = true
44
+ @output = value
45
+ end
46
+ end
47
+ end
data/lib/data-forest/node.rb ADDED
@@ -0,0 +1,45 @@
1
+ # frozen_string_literal: true
2
+
3
+ module DataForest
4
+ class Node
5
+ attr_reader :dependencies, :job_class, :name
6
+
7
+ def initialize(name, job_class, dependencies)
8
+ validate_name!(name)
9
+ validate_job_class!(job_class)
10
+ validate_dependencies!(dependencies)
11
+
12
+ @name = name
13
+ @job_class = job_class
14
+ @dependencies = dependencies
15
+ end
16
+
17
+ private
18
+
19
+ def validate_name!(name)
20
+ error = ArgumentError
21
+ message = 'name is not a Symbol'
22
+
23
+ raise(error, message) unless name.is_a?(Symbol)
24
+ end
25
+
26
+ def validate_job_class!(job)
27
+ error = ArgumentError
28
+ message = "node value: #{job} is not a DataForest::Job"
29
+
30
+ valid = job.is_a?(Class) && job.ancestors.include?(DataForest::Job)
31
+ raise(error, message) unless valid
32
+ end
33
+
34
+ def validate_dependencies!(dependencies)
35
+ error = ArgumentError
36
+ message = "dependencies: #{dependencies} is not an Array"
37
+ raise(error, message) unless dependencies.is_a?(Array)
38
+
39
+ dependencies.each do |dependency|
40
+ message = "dependencies value: #{dependency} is not a Symbol"
41
+ raise(error, message) unless dependency.is_a?(Symbol)
42
+ end
43
+ end
44
+ end
45
+ end
data/lib/data-forest/runner.rb ADDED
@@ -0,0 +1,72 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'data-forest/cache/filesystem'
4
+ require 'data-forest/cache/memory'
5
+ require 'data-forest/forest'
6
+ require 'data-forest/node'
7
+ require 'data-forest/tree'
8
+
9
+ require 'json'
10
+
11
+ module DataForest
12
+ class Runner
13
+ def call(forest, tree_name)
14
+ tree = forest.trees[tree_name]
15
+ raise ArgumentError unless tree
16
+
17
+ plan = extract_execution_plan(tree.root, tree.nodes)
18
+ execute(forest, tree, plan)
19
+ end
20
+
21
+ private
22
+
23
+ def extract_execution_plan(root, nodes)
24
+ sources = nodes[root].dependencies.dup
25
+ plan = [[root, nil, sources]]
26
+ queue = [root]
27
+
28
+ while queue.any?
29
+ name = queue.shift
30
+ current = nodes[name]
31
+ dependencies = current.dependencies
32
+
33
+ data_targets = []
34
+ if dependencies
35
+ dependencies.each do |d|
36
+ sources = nodes[d].dependencies.dup
37
+
38
+ queue.push(d)
39
+ plan.push([d, name, sources])
40
+ end
41
+ end
42
+ end
43
+
44
+ plan.reverse
45
+ end
46
+
47
+ def execute(forest, tree, plan)
48
+ cache = Cache::Filesystem.new(forest.name, tree.name)
49
+ cache.setup!
50
+
51
+ plan.each do |name, _target, sources|
52
+ next if cache.key?(name)
53
+
54
+ node = tree.nodes[name]
55
+ job_class = node.job_class
56
+ job = job_class.new(forest.name, tree.name)
57
+
58
+ if sources
59
+ sources.each do |source|
60
+ job.add_input(source, JSON.parse(cache.find!(source)))
61
+ end
62
+ end
63
+ job.perform
64
+
65
+ if job.output?
66
+ output = job.retrieve_output.to_json
67
+ cache.store!(name, output)
68
+ end
69
+ end
70
+ end
71
+ end
72
+ end
data/lib/data-forest/tree.rb ADDED
@@ -0,0 +1,53 @@
1
+ # frozen_string_literal: true
2
+
3
+ module DataForest
4
+ class Tree
5
+ attr_reader :name, :nodes, :root
6
+
7
+ def initialize(name, root, nodes)
8
+ validate_name!(name)
9
+ validate_nodes!(nodes)
10
+ validate_root!(root, nodes)
11
+
12
+ @name = name
13
+ @nodes = nodes
14
+ @root = root
15
+ end
16
+
17
+ private
18
+
19
+ def validate_name!(name)
20
+ error = ArgumentError
21
+ message = 'name is not a Symbol'
22
+
23
+ raise(error, message) unless name.is_a?(Symbol)
24
+ end
25
+
26
+ def validate_nodes!(nodes)
27
+ error = ArgumentError
28
+ message = 'nodes is not a Hash'
29
+
30
+ raise(error, message) unless nodes.is_a?(Hash)
31
+
32
+ nodes.each do |key, value|
33
+ message = "node key: #{key} is not a Symbol, but a #{key.class.name}"
34
+ raise(error, message) unless key.is_a?(Symbol)
35
+
36
+ message = "node value: #{value} is not a DataForest::Node, "\
37
+ "but a #{value.class.name}"
38
+ raise(error, message) unless value.is_a?(DataForest::Node)
39
+ end
40
+ end
41
+
42
+ def validate_root!(root, nodes)
43
+ return if root.nil? && nodes.empty?
44
+
45
+ error = ArgumentError
46
+ message = "root value: #{root} is not a Symbol, but a #{root.class.name}"
47
+ raise(error, message) unless root.is_a?(Symbol)
48
+
49
+ message = "root value: #{root} is not a key of the `nodes` hash"
50
+ raise(error, message) unless nodes.key?(root)
51
+ end
52
+ end
53
+ end
metadata ADDED
@@ -0,0 +1,180 @@
1
+ --- !ruby/object:Gem::Specification
2
+ name: data-forest
3
+ version: !ruby/object:Gem::Version
4
+ version: 0.0.2
5
+ platform: ruby
6
+ authors:
7
+ - Codruț Constantin Gușoi
8
+ autorequire:
9
+ bindir: bin
10
+ cert_chain: []
11
+ date: 2019-12-02 00:00:00.000000000 Z
12
+ dependencies:
13
+ - !ruby/object:Gem::Dependency
14
+ name: bundler
15
+ requirement: !ruby/object:Gem::Requirement
16
+ requirements:
17
+ - - "~>"
18
+ - !ruby/object:Gem::Version
19
+ version: '2.0'
20
+ type: :development
21
+ prerelease: false
22
+ version_requirements: !ruby/object:Gem::Requirement
23
+ requirements:
24
+ - - "~>"
25
+ - !ruby/object:Gem::Version
26
+ version: '2.0'
27
+ - !ruby/object:Gem::Dependency
28
+ name: irb
29
+ requirement: !ruby/object:Gem::Requirement
30
+ requirements:
31
+ - - "~>"
32
+ - !ruby/object:Gem::Version
33
+ version: 1.1.0
34
+ type: :development
35
+ prerelease: false
36
+ version_requirements: !ruby/object:Gem::Requirement
37
+ requirements:
38
+ - - "~>"
39
+ - !ruby/object:Gem::Version
40
+ version: 1.1.0
41
+ - !ruby/object:Gem::Dependency
42
+ name: minitest
43
+ requirement: !ruby/object:Gem::Requirement
44
+ requirements:
45
+ - - "~>"
46
+ - !ruby/object:Gem::Version
47
+ version: '5.0'
48
+ type: :development
49
+ prerelease: false
50
+ version_requirements: !ruby/object:Gem::Requirement
51
+ requirements:
52
+ - - "~>"
53
+ - !ruby/object:Gem::Version
54
+ version: '5.0'
55
+ - !ruby/object:Gem::Dependency
56
+ name: pry-byebug
57
+ requirement: !ruby/object:Gem::Requirement
58
+ requirements:
59
+ - - "~>"
60
+ - !ruby/object:Gem::Version
61
+ version: 3.7.0
62
+ type: :development
63
+ prerelease: false
64
+ version_requirements: !ruby/object:Gem::Requirement
65
+ requirements:
66
+ - - "~>"
67
+ - !ruby/object:Gem::Version
68
+ version: 3.7.0
69
+ - !ruby/object:Gem::Dependency
70
+ name: pry-doc
71
+ requirement: !ruby/object:Gem::Requirement
72
+ requirements:
73
+ - - "~>"
74
+ - !ruby/object:Gem::Version
75
+ version: 1.0.0
76
+ type: :development
77
+ prerelease: false
78
+ version_requirements: !ruby/object:Gem::Requirement
79
+ requirements:
80
+ - - "~>"
81
+ - !ruby/object:Gem::Version
82
+ version: 1.0.0
83
+ - !ruby/object:Gem::Dependency
84
+ name: rake
85
+ requirement: !ruby/object:Gem::Requirement
86
+ requirements:
87
+ - - "~>"
88
+ - !ruby/object:Gem::Version
89
+ version: 13.0.1
90
+ type: :development
91
+ prerelease: false
92
+ version_requirements: !ruby/object:Gem::Requirement
93
+ requirements:
94
+ - - "~>"
95
+ - !ruby/object:Gem::Version
96
+ version: 13.0.1
97
+ - !ruby/object:Gem::Dependency
98
+ name: rubocop
99
+ requirement: !ruby/object:Gem::Requirement
100
+ requirements:
101
+ - - "~>"
102
+ - !ruby/object:Gem::Version
103
+ version: 0.77.0
104
+ type: :development
105
+ prerelease: false
106
+ version_requirements: !ruby/object:Gem::Requirement
107
+ requirements:
108
+ - - "~>"
109
+ - !ruby/object:Gem::Version
110
+ version: 0.77.0
111
+ - !ruby/object:Gem::Dependency
112
+ name: solargraph
113
+ requirement: !ruby/object:Gem::Requirement
114
+ requirements:
115
+ - - "~>"
116
+ - !ruby/object:Gem::Version
117
+ version: 0.38.0
118
+ type: :development
119
+ prerelease: false
120
+ version_requirements: !ruby/object:Gem::Requirement
121
+ requirements:
122
+ - - "~>"
123
+ - !ruby/object:Gem::Version
124
+ version: 0.38.0
125
+ description:
126
+ email:
127
+ - codrut.gusoi@gmail.com
128
+ executables:
129
+ - data-forest
130
+ extensions: []
131
+ extra_rdoc_files: []
132
+ files:
133
+ - CHANGELOG.md
134
+ - Gemfile
135
+ - LICENSE
136
+ - README.md
137
+ - bin/data-forest
138
+ - data-forest.gemspec
139
+ - lib/data-forest.rb
140
+ - lib/data-forest/builder.rb
141
+ - lib/data-forest/cache/base.rb
142
+ - lib/data-forest/cache/filesystem.rb
143
+ - lib/data-forest/cache/memory.rb
144
+ - lib/data-forest/cli.rb
145
+ - lib/data-forest/cli/command.rb
146
+ - lib/data-forest/cli/command/main.rb
147
+ - lib/data-forest/cli/command/run.rb
148
+ - lib/data-forest/cli/runner.rb
149
+ - lib/data-forest/forest.rb
150
+ - lib/data-forest/job.rb
151
+ - lib/data-forest/node.rb
152
+ - lib/data-forest/runner.rb
153
+ - lib/data-forest/tree.rb
154
+ homepage: https://gitlab.com/sdwolfz/data-forest
155
+ licenses:
156
+ - BSD-3-Clause
157
+ metadata:
158
+ homepage_uri: https://gitlab.com/sdwolfz/data-forest
159
+ source_code_uri: https://gitlab.com/sdwolfz/data-forest
160
+ changelog_uri: https://gitlab.com/sdwolfz/data-forest/blob/master/CHANGELOG.md
161
+ post_install_message:
162
+ rdoc_options: []
163
+ require_paths:
164
+ - lib
165
+ required_ruby_version: !ruby/object:Gem::Requirement
166
+ requirements:
167
+ - - ">="
168
+ - !ruby/object:Gem::Version
169
+ version: '0'
170
+ required_rubygems_version: !ruby/object:Gem::Requirement
171
+ requirements:
172
+ - - ">="
173
+ - !ruby/object:Gem::Version
174
+ version: '0'
175
+ requirements: []
176
+ rubygems_version: 3.0.6
177
+ signing_key:
178
+ specification_version: 4
179
+ summary: Define and execute data processing trees.
180
+ test_files: []