summaryrefslogtreecommitdiff
path: root/vendor/bundle/ruby/3.4.0/gems/jekyll-4.4.1/lib/jekyll/readers/data_reader.rb
blob: 80b57bd6bd733451014a434d44f66458465a1a54 (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
# frozen_string_literal: true

module Jekyll
  class DataReader
    attr_reader :site, :content

    def initialize(site, in_source_dir: nil)
      @site = site
      @content = {}
      @entry_filter = EntryFilter.new(site)
      @in_source_dir = in_source_dir || @site.method(:in_source_dir)
      @source_dir = @in_source_dir.call("/")
    end

    # Read all the files in <dir> and adds them to @content
    #
    # dir - The String relative path of the directory to read.
    #
    # Returns @content, a Hash of the .yaml, .yml,
    # .json, and .csv files in the base directory
    def read(dir)
      base = @in_source_dir.call(dir)
      read_data_to(base, @content)
      @content
    end

    # Read and parse all .yaml, .yml, .json, .csv and .tsv
    # files under <dir> and add them to the <data> variable.
    #
    # dir - The string absolute path of the directory to read.
    # data - The variable to which data will be added.
    #
    # Returns nothing
    def read_data_to(dir, data)
      return unless File.directory?(dir) && !@entry_filter.symlink?(dir)

      entries = Dir.chdir(dir) do
        Dir["*.{yaml,yml,json,csv,tsv}"] + Dir["*"].select { |fn| File.directory?(fn) }
      end

      entries.each do |entry|
        path = @in_source_dir.call(dir, entry)
        next if @entry_filter.symlink?(path)

        if File.directory?(path)
          read_data_to(path, data[sanitize_filename(entry)] = {})
        else
          key = sanitize_filename(File.basename(entry, ".*"))
          data[key] = read_data_file(path)
        end
      end
    end

    # Determines how to read a data file.
    #
    # Returns the contents of the data file.
    def read_data_file(path)
      Jekyll.logger.debug "Reading:", path.sub(@source_dir, "")

      case File.extname(path).downcase
      when ".csv"
        CSV.read(path, **csv_config).map { |row| convert_row(row) }
      when ".tsv"
        CSV.read(path, **tsv_config).map { |row| convert_row(row) }
      else
        SafeYAML.load_file(path)
      end
    end

    def sanitize_filename(name)
      name.gsub(%r![^\w\s-]+|(?<=^|\b\s)\s+(?=$|\s?\b)!, "")
        .gsub(%r!\s+!, "_")
    end

    private

    # @return [Hash]
    def csv_config
      @csv_config ||= read_config("csv_reader")
    end

    # @return [Hash]
    def tsv_config
      @tsv_config ||= read_config("tsv_reader", { :col_sep => "\t" })
    end

    # @param config_key [String]
    # @param overrides [Hash]
    # @return [Hash]
    # @see https://ruby-doc.org/stdlib-2.5.0/libdoc/csv/rdoc/CSV.html#Converters
    def read_config(config_key, overrides = {})
      reader_config = config[config_key] || {}

      defaults = {
        :converters => reader_config.fetch("csv_converters", []).map(&:to_sym),
        :headers    => reader_config.fetch("headers", true),
        :encoding   => reader_config.fetch("encoding", config["encoding"]),
      }

      defaults.merge(overrides)
    end

    def config
      @config ||= site.config
    end

    # @param row [Array, CSV::Row]
    # @return [Array, Hash]
    def convert_row(row)
      row.instance_of?(CSV::Row) ? row.to_hash : row
    end
  end
end