automatic 26.08 → 26.09

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +80 -46
  3. data/VERSION +1 -1
  4. data/assets/siteinfo/items_all.json +60300 -52138
  5. data/automatic.gemspec +4 -3
  6. data/bin/automatic +1 -1
  7. data/doc/AI_TUTORIAL.md +64 -40
  8. data/doc/BASIC_DESIGN.md +31 -0
  9. data/doc/DEPLOYMENT.md +53 -47
  10. data/doc/PLUGINS.md +231 -91
  11. data/doc/POLICY.md +149 -44
  12. data/doc/QUICKSTART.md +19 -15
  13. data/doc/RELEASING.md +20 -7
  14. data/doc/REQUIREMENTS.md +40 -10
  15. data/doc/VERSIONS +112 -54
  16. data/lib/automatic/cli.rb +40 -18
  17. data/lib/automatic/environment.rb +1 -1
  18. data/lib/automatic/feed_maker.rb +57 -18
  19. data/lib/automatic/feed_parser.rb +1 -1
  20. data/lib/automatic/http.rb +1 -1
  21. data/lib/automatic/log.rb +1 -1
  22. data/lib/automatic/pipeline.rb +15 -5
  23. data/lib/automatic/recipe.rb +45 -1
  24. data/lib/automatic/version.rb +3 -3
  25. data/lib/automatic.rb +18 -3
  26. data/plugins/custom_feed/web.rb +1 -1
  27. data/plugins/filter/absolute_uri.rb +1 -1
  28. data/plugins/filter/batch.rb +97 -0
  29. data/plugins/filter/claude.rb +1 -1
  30. data/plugins/filter/clear.rb +1 -1
  31. data/plugins/filter/description_link.rb +20 -3
  32. data/plugins/filter/full_feed.rb +28 -16
  33. data/plugins/filter/gemini.rb +1 -1
  34. data/plugins/filter/ignore.rb +1 -1
  35. data/plugins/filter/image.rb +1 -1
  36. data/plugins/filter/image_source.rb +20 -3
  37. data/plugins/filter/join.rb +4 -6
  38. data/plugins/filter/kimi.rb +216 -0
  39. data/plugins/filter/limit.rb +57 -0
  40. data/plugins/filter/open_ai.rb +1 -1
  41. data/plugins/filter/present.rb +75 -0
  42. data/plugins/filter/sakura_ai.rb +1 -1
  43. data/plugins/filter/sanitize.rb +1 -1
  44. data/plugins/filter/sort.rb +1 -1
  45. data/plugins/filter/tumblr_resize.rb +1 -1
  46. data/plugins/notify/ikachan.rb +1 -1
  47. data/plugins/provide/fluentd.rb +1 -1
  48. data/plugins/publish/amazon_s3.rb +9 -3
  49. data/plugins/publish/console.rb +1 -1
  50. data/plugins/publish/fluentd.rb +1 -1
  51. data/plugins/publish/hatena_bookmark.rb +1 -1
  52. data/plugins/publish/markdown.rb +1 -1
  53. data/plugins/publish/memcached.rb +1 -1
  54. data/plugins/store/digest.rb +1 -1
  55. data/plugins/store/file.rb +11 -3
  56. data/plugins/store/full_text.rb +9 -13
  57. data/plugins/store/permalink.rb +1 -1
  58. data/plugins/subscription/feed.rb +1 -1
  59. data/plugins/subscription/link.rb +1 -1
  60. data/plugins/subscription/text.rb +13 -4
  61. data/plugins/subscription/tumblr.rb +1 -1
  62. data/plugins/subscription/xml.rb +1 -1
  63. metadata +6 -2
@@ -1,16 +1,16 @@
1
1
  # -*- coding: utf-8 -*-
2
2
  # Name:: Automatic::VERSION
3
- # Author: id774 (More info: http://id774.net)
3
+ # Author: id774 (More info: https://id774.net)
4
4
  # Source Code:: https://github.com/id774/automaticruby
5
5
  # License:: The GPL version 3, or LGPL version 3 (Dual License).
6
6
  # Contact:: idnanashi@gmail.com
7
7
  # Created:: Feb 18, 2012
8
- # Updated:: Aug 14, 2026
8
+ # Updated:: Aug 25, 2026
9
9
  # Copyright:: Copyright (c) 2012-2026 Automatic Ruby Developers.
10
10
 
11
11
  module Automatic
12
12
  # The release version. The VERSION file at the repository root carries the
13
13
  # same number and the gemspec reads it from there; a spec asserts that the
14
14
  # two agree. See doc/POLICY.md section 10.
15
- VERSION = '26.08'
15
+ VERSION = '26.09'
16
16
  end
data/lib/automatic.rb CHANGED
@@ -1,11 +1,11 @@
1
1
  # -*- coding: utf-8 -*-
2
2
  # Name:: Automatic::Ruby
3
- # Author: id774 (More info: http://id774.net)
3
+ # Author: id774 (More info: https://id774.net)
4
4
  # Source Code:: https://github.com/id774/automaticruby
5
5
  # License:: The GPL version 3, or LGPL version 3 (Dual License).
6
6
  # Contact:: idnanashi@gmail.com
7
7
  # Created:: Feb 18, 2012
8
- # Updated:: Aug 14, 2026
8
+ # Updated:: Sep 6, 2026
9
9
  # Copyright:: Copyright (c) 2012-2026 Automatic Ruby Developers.
10
10
  #
11
11
  # The framework module: the two directories everything else resolves paths
@@ -38,6 +38,14 @@ module Automatic
38
38
  # A Recipe parsed, but is not a document this framework can run.
39
39
  class InvalidRecipeError < Error; end
40
40
 
41
+ # Raised by require_optional when, and only when, the exact feature it was
42
+ # asked to require could not be found. A LoadError subtype rather than an
43
+ # Error subtype, so that a caller already rescuing LoadError still catches
44
+ # it; distinct from plain LoadError so that a caller can tell "the optional
45
+ # gem itself is missing" apart from a LoadError raised from inside that
46
+ # gem's own load. See doc/POLICY.md section 9.1.
47
+ class OptionalDependencyError < LoadError; end
48
+
41
49
  class << self
42
50
  attr_accessor :root_dir
43
51
 
@@ -55,7 +63,14 @@ module Automatic
55
63
  def require_optional(feature, needed_by:, gem_name: feature)
56
64
  require feature
57
65
  rescue LoadError => e
58
- raise LoadError,
66
+ # e.path is the argument require failed to find. When it matches
67
+ # feature, this require itself is what failed, and the gem naming this
68
+ # feature is what is missing. When it does not, the failure happened
69
+ # somewhere inside feature's own load -- a different missing file -- and
70
+ # is not this gem's absence; it is re-raised unconverted.
71
+ raise e unless e.path == feature
72
+
73
+ raise OptionalDependencyError,
59
74
  "The `#{gem_name}` gem is not installed. It is needed by #{needed_by}. " \
60
75
  "Install it with `gem install #{gem_name}`, or in a source checkout add " \
61
76
  'its group to the bundle; see the optional plugin dependencies in ' \
@@ -1,7 +1,7 @@
1
1
  # -*- coding: utf-8 -*-
2
2
  # Name:: Automatic::Plugin::CustomFeed::Web
3
3
  # Description:: Build a feed from the article links of an HTML index page.
4
- # Author: id774 (More info: http://id774.net)
4
+ # Author: id774 (More info: https://id774.net)
5
5
  # Source Code:: https://github.com/id774/automaticruby
6
6
  # License:: The GPL version 3, or LGPL version 3 (Dual License).
7
7
  # Contact:: idnanashi@gmail.com
@@ -1,6 +1,6 @@
1
1
  # -*- coding: utf-8 -*-
2
2
  # Name:: Automatic::Plugin::Filter::AbsoluteURI
3
- # Author: id774 (More info: http://id774.net)
3
+ # Author: id774 (More info: https://id774.net)
4
4
  # Source Code:: https://github.com/id774/automaticruby
5
5
  # License:: The GPL version 3, or LGPL version 3 (Dual License).
6
6
  # Contact:: idnanashi@gmail.com
@@ -0,0 +1,97 @@
1
+ # -*- coding: utf-8 -*-
2
+ # Name:: Automatic::Plugin::Filter::Batch
3
+ # Description:: Group the whole pipeline into fixed-size item batches.
4
+ # Author: id774 (More info: https://id774.net)
5
+ # Source Code:: https://github.com/id774/automaticruby
6
+ # License:: The GPL version 3, or LGPL version 3 (Dual License).
7
+ # Contact:: idnanashi@gmail.com
8
+ # Created:: Aug 24, 2026
9
+ # Updated:: Aug 24, 2026
10
+ # Copyright:: Copyright (c) 2012-2026 Automatic Ruby Developers.
11
+
12
+ module Automatic::Plugin
13
+ class FilterBatch
14
+ require 'rss'
15
+
16
+ # Where one source item ends and the next begins inside a batch's
17
+ # description, in the manner of FilterJoin's own delimiter -- but built
18
+ # independently, since a batch item is not a joined item.
19
+ HEADING = 'ARTICLE'.freeze
20
+
21
+ def initialize(config, pipeline = [])
22
+ @config = config || {}
23
+ @pipeline = pipeline
24
+ @batch_items = validated_batch_items
25
+ end
26
+
27
+ # Collects the whole pipeline into one Array, discarding feed boundaries,
28
+ # and slices it into fixed-size batches. Each batch becomes one item in one
29
+ # output feed.
30
+ def run
31
+ items = collect
32
+ return [] if items.empty?
33
+
34
+ feed(items.each_slice(@batch_items).to_a)
35
+ end
36
+
37
+ private
38
+
39
+ def validated_batch_items
40
+ value = begin
41
+ Integer(@config['batch_items'].to_s, 10)
42
+ rescue ArgumentError
43
+ nil
44
+ end
45
+
46
+ if value.nil? || value < 1
47
+ raise ArgumentError, 'FilterBatch needs batch_items to be a positive integer'
48
+ end
49
+
50
+ value
51
+ end
52
+
53
+ def collect
54
+ @pipeline.each_with_object([]) do |feeds, items|
55
+ next if feeds.nil?
56
+
57
+ items.concat(feeds.items)
58
+ end
59
+ end
60
+
61
+ def feed(batches)
62
+ [RSS::Maker.make('2.0') { |maker|
63
+ maker.channel.title = 'Automatic Ruby'
64
+ maker.channel.description = 'Automatic::Plugin::FilterBatch'
65
+ maker.channel.link = 'https://github.com/id774/automaticruby'
66
+ maker.items.do_sort = false
67
+
68
+ batches.each_with_index do |batch, index|
69
+ item = maker.items.new_item
70
+ item.title = "Batch #{index + 1}"
71
+ item.description = description(batch)
72
+ item.date = Time.now
73
+ end
74
+ }]
75
+ end
76
+
77
+ def description(batch)
78
+ batch.each_with_index.map { |item, index| section(index + 1, item) }.join("\n\n")
79
+ end
80
+
81
+ def section(number, item)
82
+ ["#{HEADING} #{number}",
83
+ "Title: #{value(item, :title)}",
84
+ "URL: #{value(item, :link)}",
85
+ '',
86
+ value(item, :description)].join("\n")
87
+ end
88
+
89
+ # A field an item does not carry is empty rather than absent, so that every
90
+ # ARTICLE has the same shape whatever the feed it came from left out.
91
+ def value(item, name)
92
+ return '' unless item.respond_to?(name)
93
+
94
+ item.public_send(name).to_s.strip
95
+ end
96
+ end
97
+ end
@@ -1,7 +1,7 @@
1
1
  # -*- coding: utf-8 -*-
2
2
  # Name:: Automatic::Plugin::Filter::Claude
3
3
  # Description:: Replace each item's description with what the Anthropic Claude API answers.
4
- # Author: id774 (More info: http://id774.net)
4
+ # Author: id774 (More info: https://id774.net)
5
5
  # Source Code:: https://github.com/id774/automaticruby
6
6
  # License:: The GPL version 3, or LGPL version 3 (Dual License).
7
7
  # Contact:: idnanashi@gmail.com
@@ -1,6 +1,6 @@
1
1
  # -*- coding: utf-8 -*-
2
2
  # Name:: Automatic::Plugin::Filter::Clear
3
- # Author: id774 (More info: http://id774.net)
3
+ # Author: id774 (More info: https://id774.net)
4
4
  # Source Code:: https://github.com/id774/automaticruby
5
5
  # License:: The GPL version 3, or LGPL version 3 (Dual License).
6
6
  # Contact:: idnanashi@gmail.com
@@ -1,11 +1,11 @@
1
1
  # -*- coding: utf-8 -*-
2
2
  # Name:: Automatic::Plugin::Filter::DescriptionLink
3
- # Author: id774 (More info: http://id774.net)
3
+ # Author: id774 (More info: https://id774.net)
4
4
  # Source Code:: https://github.com/id774/automaticruby
5
5
  # License:: The GPL version 3, or LGPL version 3 (Dual License).
6
6
  # Contact:: idnanashi@gmail.com
7
7
  # Created:: Oct 03, 2014
8
- # Updated:: Aug 15, 2026
8
+ # Updated:: Aug 24, 2026
9
9
  # Copyright:: Copyright (c) 2012-2026 Automatic Ruby Developers.
10
10
 
11
11
  module Automatic::Plugin
@@ -63,10 +63,27 @@ module Automatic::Plugin
63
63
  def fetch_title(url)
64
64
  return nil unless Automatic::Http.fetchable?(url)
65
65
 
66
- Nokogiri::HTML.parse(Automatic::Http.read(url)).xpath('//title').text
66
+ Nokogiri::HTML.parse(page(url)).xpath('//title').text
67
67
  rescue StandardError => e
68
68
  Automatic::Log.puts('warn', "Failed in get title for: #{url}, #{e.message}")
69
69
  nil
70
70
  end
71
+
72
+ # The one place this plugin actually reaches the network. `wait` runs in
73
+ # the `ensure` so that a fetch attempt is waited out whether it succeeded
74
+ # or raised -- a URL that is not fetchable never gets here at all, and so
75
+ # never waits.
76
+ def page(url)
77
+ Automatic::Http.read(url)
78
+ ensure
79
+ wait
80
+ end
81
+
82
+ # `interval` seconds after a real fetch attempt, positive values only. See
83
+ # doc/PLUGINS.md section 6.3.
84
+ def wait
85
+ seconds = @config['interval'].to_i
86
+ sleep(seconds) if seconds.positive?
87
+ end
71
88
  end
72
89
  end
@@ -5,7 +5,7 @@
5
5
  # License:: The GPL version 3, or LGPL version 3 (Dual License).
6
6
  # Contact:: idnanashi@gmail.com
7
7
  # Created:: Apr 29, 2012
8
- # Updated:: Aug 15, 2026
8
+ # Updated:: Aug 24, 2026
9
9
  # Copyright:: Copyright (c) 2012-2026 Automatic Ruby Developers.
10
10
 
11
11
  module Automatic::Plugin
@@ -17,10 +17,8 @@ module Automatic::Plugin
17
17
  SITEINFO_TYPES = %w[SBM INDIVIDUAL IND SUBGENERAL SUB GENERAL GEN].freeze
18
18
 
19
19
  # One siteinfo record, reduced to the four things a match needs and with
20
- # its URL pattern compiled once. The database ships with 3,504 usable
21
- # records, so compiling them per item -- which is what matching against
22
- # the raw JSON did -- was several thousand `Regexp.new` calls for every
23
- # link in every feed.
20
+ # its URL pattern compiled once. Compiling patterns when the database is
21
+ # loaded avoids rebuilding regular expressions for every item.
24
22
  Entry = Struct.new(:url, :pattern, :xpath, :encoding)
25
23
 
26
24
  def initialize(config, pipeline = [])
@@ -107,13 +105,10 @@ module Automatic::Plugin
107
105
  @siteinfo.find { |record| links.any? { |candidate| record.pattern.match?(candidate) } }
108
106
  end
109
107
 
110
- # The database was last updated in 2013 and 3,448 of its 3,504 records
111
- # anchor on a scheme, nearly all of them `^http://`. The sites they name
112
- # have since moved to HTTPS, so an https link out of a feed matches none of
113
- # them and the filter silently does nothing. Matching the link under either
114
- # scheme is what keeps those records reachable; a record is about a site's
115
- # layout, not about how it is transported. Only the match is rewritten --
116
- # the page is fetched from the link the feed gave.
108
+ # A stored siteinfo pattern may name HTTP while a current feed supplies HTTPS,
109
+ # or the reverse. A record describes a site's layout rather than its current
110
+ # transport scheme, so matching tries both schemes. Only the candidate used
111
+ # for matching changes; the page is fetched from the original item link.
117
112
  def schemes(link)
118
113
  case link
119
114
  when %r{\Ahttps://} then [link, link.sub(%r{\Ahttps://}, 'http://')]
@@ -150,17 +145,34 @@ module Automatic::Plugin
150
145
  # database is full of sites that declare their charset only in a meta tag,
151
146
  # and for those the difference is the whole article in mojibake.
152
147
  #
153
- # A record's own `enc` is the last resort, for a page that declares nothing
154
- # anywhere: it was recorded in 2013, and trusting it ahead of what the page
155
- # says would break every site that has changed encoding since.
148
+ # A record's own `enc` is the last resort for a page that declares nothing.
149
+ # Page declarations take precedence because a site's encoding can change after
150
+ # a siteinfo record is written.
156
151
  def document(link, entry)
157
- page, declared = Automatic::Http.open(link) { |io| [io.read, declared_charset?(io)] }
152
+ page, declared = fetch_page(link)
158
153
  parsed = Nokogiri::HTML.parse(StringIO.new(page))
159
154
  return parsed if declared || parsed.meta_encoding || entry.encoding.nil?
160
155
 
161
156
  Nokogiri::HTML.parse(StringIO.new(page), nil, entry.encoding)
162
157
  end
163
158
 
159
+ # The one place this plugin actually reaches the network. `wait` runs in
160
+ # the `ensure` so that a fetch attempt is waited out whether it succeeded
161
+ # or raised -- an item with no link and an item whose siteinfo did not
162
+ # match never get here at all, and so never wait.
163
+ def fetch_page(link)
164
+ Automatic::Http.open(link) { |io| [io.read, declared_charset?(io)] }
165
+ ensure
166
+ wait
167
+ end
168
+
169
+ # `interval` seconds after a real fetch attempt, positive values only. See
170
+ # doc/PLUGINS.md section 6.3.
171
+ def wait
172
+ seconds = @config['interval'].to_i
173
+ sleep(seconds) if seconds.positive?
174
+ end
175
+
164
176
  # Whether the response itself named a charset, as opposed to open-uri
165
177
  # having settled on one in the absence of an answer.
166
178
  def declared_charset?(io)
@@ -1,7 +1,7 @@
1
1
  # -*- coding: utf-8 -*-
2
2
  # Name:: Automatic::Plugin::Filter::Gemini
3
3
  # Description:: Replace each item's description with what the Google Gemini API answers.
4
- # Author: id774 (More info: http://id774.net)
4
+ # Author: id774 (More info: https://id774.net)
5
5
  # Source Code:: https://github.com/id774/automaticruby
6
6
  # License:: The GPL version 3, or LGPL version 3 (Dual License).
7
7
  # Contact:: idnanashi@gmail.com
@@ -1,6 +1,6 @@
1
1
  # -*- coding: utf-8 -*-
2
2
  # Name:: Automatic::Plugin::Filter::Ignore
3
- # Author: id774 (More info: http://id774.net)
3
+ # Author: id774 (More info: https://id774.net)
4
4
  # Source Code:: https://github.com/id774/automaticruby
5
5
  # License:: The GPL version 3, or LGPL version 3 (Dual License).
6
6
  # Contact:: idnanashi@gmail.com
@@ -1,6 +1,6 @@
1
1
  # -*- coding: utf-8 -*-
2
2
  # Name:: Automatic::Plugin::Filter::Image
3
- # Author: id774 (More info: http://id774.net)
3
+ # Author: id774 (More info: https://id774.net)
4
4
  # Source Code:: https://github.com/id774/automaticruby
5
5
  # License:: The GPL version 3, or LGPL version 3 (Dual License).
6
6
  # Contact:: idnanashi@gmail.com
@@ -1,11 +1,11 @@
1
1
  # -*- coding: utf-8 -*-
2
2
  # Name:: Automatic::Plugin::Filter::ImageSource
3
- # Author: id774 (More info: http://id774.net)
3
+ # Author: id774 (More info: https://id774.net)
4
4
  # Source Code:: https://github.com/id774/automaticruby
5
5
  # License:: The GPL version 3, or LGPL version 3 (Dual License).
6
6
  # Contact:: idnanashi@gmail.com
7
7
  # Created:: Feb 28, 2012
8
- # Updated:: Aug 15, 2026
8
+ # Updated:: Aug 24, 2026
9
9
  # Copyright:: Copyright (c) 2012-2026 Automatic Ruby Developers.
10
10
 
11
11
  module Automatic::Plugin
@@ -49,12 +49,29 @@ module Automatic::Plugin
49
49
  end
50
50
 
51
51
  def page_images(link)
52
- sources(Automatic::Http.read(link), link)
52
+ sources(page(link), link)
53
53
  rescue StandardError => e
54
54
  Automatic::Log.puts('warn', "Failed to read images from #{link}: #{e.message}")
55
55
  []
56
56
  end
57
57
 
58
+ # The one place this plugin actually reaches the network. `wait` runs in
59
+ # the `ensure` so that a fetch attempt is waited out whether it succeeded
60
+ # or raised -- an item whose description already had images never gets
61
+ # here at all, and so never waits.
62
+ def page(link)
63
+ Automatic::Http.read(link)
64
+ ensure
65
+ wait
66
+ end
67
+
68
+ # `interval` seconds after a real fetch attempt, positive values only. See
69
+ # doc/PLUGINS.md section 6.3.
70
+ def wait
71
+ seconds = @config['interval'].to_i
72
+ sleep(seconds) if seconds.positive?
73
+ end
74
+
58
75
  # The images of an HTML fragment, as absolute URLs. This was a scan for
59
76
  # `<img src="` before, which found nothing in a document quoting its
60
77
  # attributes with apostrophes or writing src after another attribute; the
@@ -1,7 +1,7 @@
1
1
  # -*- coding: utf-8 -*-
2
2
  # Name:: Automatic::Plugin::Filter::Join
3
3
  # Description:: Join every item in the pipeline into one item.
4
- # Author: id774 (More info: http://id774.net)
4
+ # Author: id774 (More info: https://id774.net)
5
5
  # Source Code:: https://github.com/id774/automaticruby
6
6
  # License:: The GPL version 3, or LGPL version 3 (Dual License).
7
7
  # Contact:: idnanashi@gmail.com
@@ -87,11 +87,9 @@ module Automatic::Plugin
87
87
  item.public_send(name).to_s.strip
88
88
  end
89
89
 
90
- # Built here rather than through FeedMaker.create_pipeline, which drops an
91
- # item whose link is nil -- and this item's link is nil deliberately. It is
92
- # several articles at once, so there is no page it points at, and putting
93
- # the first article's URL there would name a source for text that is not
94
- # only from it.
90
+ # This joined item deliberately has no link. It is several articles at once,
91
+ # so there is no single page it points at; using the first article's URL would
92
+ # name a source for text that is not only from it.
95
93
  def feed(item_title, item_description)
96
94
  RSS::Maker.make('2.0') { |maker|
97
95
  maker.channel.title = 'Automatic Ruby'
@@ -0,0 +1,216 @@
1
+ # -*- coding: utf-8 -*-
2
+ # Name:: Automatic::Plugin::Filter::Kimi
3
+ # Description:: Replace each item's description with what the Kimi API answers.
4
+ # Author: id774 (More info: https://id774.net)
5
+ # Source Code:: https://github.com/id774/automaticruby
6
+ # License:: The GPL version 3, or LGPL version 3 (Dual License).
7
+ # Contact:: idnanashi@gmail.com
8
+ # Created:: Sep 11, 2026
9
+ # Updated:: Sep 11, 2026
10
+ # Copyright:: Copyright (c) 2012-2026 Automatic Ruby Developers.
11
+ #
12
+ # One transformation: the item's description goes to the Kimi API under the
13
+ # Recipe's prompt, and the answer becomes the item's description. What that
14
+ # transformation is -- a summary, a translation, an extraction, a
15
+ # classification -- is the prompt's business, not this plugin's.
16
+ #
17
+ # Moonshot AI's Kimi offers an OpenAI-compatible chat completions interface,
18
+ # and this plugin is still its own rather than a mode of FilterOpenAI. It is a
19
+ # different service: a different endpoint, a different account, a different
20
+ # set of models, its own limits and its own errors, and any of those may move
21
+ # without OpenAI moving. A Recipe naming FilterKimi says which service the
22
+ # text is sent to, which a `provider:` setting would not.
23
+ #
24
+ # Kimi may return a `reasoning_content` alongside the answer's own `content`.
25
+ # That reasoning is never read, logged or written anywhere by this plugin --
26
+ # only `content` is the answer, and only once `finish_reason` says the model
27
+ # is done.
28
+ #
29
+ # @see https://platform.moonshot.ai/docs/api/chat
30
+
31
+ module Automatic::Plugin
32
+ class FilterKimi
33
+ require 'json'
34
+ require 'net/http'
35
+ require 'openssl'
36
+ require 'uri'
37
+
38
+ # The one endpoint the service publishes for this. It is not a setting: an
39
+ # operator has no version of this plugin that talks to a different host,
40
+ # and a setting for it would be a way to send the token somewhere else.
41
+ ENDPOINT = URI('https://api.moonshot.ai/v1/chat/completions')
42
+
43
+ OPEN_TIMEOUT = 10
44
+
45
+ # Generous, and bounded. A model given several articles thinks for a while;
46
+ # an unattended run that waits forever is the failure this exists against.
47
+ READ_TIMEOUT = 300
48
+
49
+ # A failure that another attempt will not get past: a setting that is
50
+ # wrong, a request the service refuses, an answer this plugin cannot read.
51
+ class Error < StandardError; end
52
+
53
+ # A failure that another attempt may get past: the network, a rate limit, a
54
+ # server error.
55
+ class TemporaryError < StandardError; end
56
+
57
+ def initialize(config, pipeline = [])
58
+ @config = config || {}
59
+ @pipeline = pipeline
60
+ end
61
+
62
+ # Replaces each item's description with the answer. Nothing else about an
63
+ # item is touched, and the feeds and their items arrive and leave in the
64
+ # same number and the same order.
65
+ def run
66
+ validate_settings
67
+
68
+ @pipeline.each { |feeds|
69
+ next if feeds.nil?
70
+
71
+ feeds.items.each { |item| transform(item) }
72
+ }
73
+ @pipeline
74
+ end
75
+
76
+ private
77
+
78
+ # Checked before the first request, because a Recipe this plugin cannot
79
+ # carry out is the operator's mistake and will be the same mistake on every
80
+ # item. The token is never named in a message.
81
+ def validate_settings
82
+ raise ArgumentError, 'FilterKimi needs a token' if token.empty?
83
+ raise ArgumentError, 'FilterKimi needs a model' if model.empty?
84
+ raise ArgumentError, 'FilterKimi needs a prompt' if prompt.empty?
85
+ end
86
+
87
+ def token
88
+ @config['token'].to_s
89
+ end
90
+
91
+ def model
92
+ @config['model'].to_s.strip
93
+ end
94
+
95
+ def prompt
96
+ @config['prompt'].to_s.strip
97
+ end
98
+
99
+ def transform(item)
100
+ text = item.description.to_s
101
+ if text.strip.empty?
102
+ Automatic::Log.puts('warn', "FilterKimi: nothing to send for #{item.link}")
103
+ return
104
+ end
105
+
106
+ Automatic::Log.puts('info', "FilterKimi: asking #{model} about #{item.link}")
107
+ item.description = answer(text)
108
+ end
109
+
110
+ # The retry shape of doc/PLUGINS.md section 3.6, applied only to what
111
+ # retrying can help. A missing setting, a refused request or an answer in a
112
+ # shape this plugin cannot read is raised at once: trying again would fail
113
+ # the same way, more slowly.
114
+ def answer(text)
115
+ retries = 0
116
+ retry_max = @config['retry'].to_i
117
+ begin
118
+ completion(text)
119
+ rescue TemporaryError => e
120
+ retries += 1
121
+ Automatic::Log.puts('error', "ErrorCount: #{retries}, FilterKimi: #{e.message}")
122
+ if retries <= retry_max
123
+ sleep(@config['interval'].to_i)
124
+ retry
125
+ end
126
+ raise Error, "FilterKimi gave up after #{retries} attempts: #{e.message}"
127
+ end
128
+ end
129
+
130
+ def completion(text)
131
+ # The prompt is the system turn and the description is the user turn it
132
+ # is applied to. They are separate messages, so that what an article says
133
+ # is never read as an instruction to this plugin or to the model. Nothing
134
+ # else -- no sampling or tool-use parameter -- is sent.
135
+ body = {
136
+ 'model' => model,
137
+ 'messages' => [
138
+ { 'role' => 'system', 'content' => prompt },
139
+ { 'role' => 'user', 'content' => text }
140
+ ]
141
+ }
142
+ content(post(JSON.generate(body)))
143
+ end
144
+
145
+ def post(body)
146
+ request = Net::HTTP::Post.new(ENDPOINT)
147
+ request['Authorization'] = "Bearer #{token}"
148
+ request['Content-Type'] = 'application/json'
149
+ request.body = body
150
+
151
+ # TLS with the certificate verified, which is Net::HTTP's own default and
152
+ # is named here because it is not a thing to be turned off.
153
+ Net::HTTP.start(ENDPOINT.host, ENDPOINT.port,
154
+ use_ssl: true,
155
+ verify_mode: OpenSSL::SSL::VERIFY_PEER,
156
+ open_timeout: OPEN_TIMEOUT,
157
+ read_timeout: READ_TIMEOUT) { |http| http.request(request) }
158
+ rescue Timeout::Error, SystemCallError, SocketError, IOError,
159
+ OpenSSL::SSL::SSLError, Net::HTTPBadResponse => e
160
+ raise TemporaryError, "the request to Kimi failed: #{e.message}"
161
+ end
162
+
163
+ def content(response)
164
+ case response
165
+ when Net::HTTPSuccess
166
+ answer_text(parse(response.body))
167
+ when Net::HTTPTooManyRequests, Net::HTTPServerError
168
+ raise TemporaryError, "Kimi answered #{response.code}: #{reason(response)}"
169
+ else
170
+ raise Error, "Kimi answered #{response.code}: #{reason(response)}"
171
+ end
172
+ end
173
+
174
+ def parse(body)
175
+ JSON.parse(body.to_s)
176
+ rescue JSON::ParserError => e
177
+ raise Error, "Kimi answered with something that is not JSON: #{e.message}"
178
+ end
179
+
180
+ # The first choice's message content, once the model says it is actually
181
+ # done. `reasoning_content`, when Kimi sends one alongside `content`, is
182
+ # never read here: it is not the answer, and it never becomes one. An
183
+ # answer this plugin cannot find is an error and not an empty description:
184
+ # a Recipe that published the empty string here would have thrown the
185
+ # article away and reported success.
186
+ def answer_text(body)
187
+ choices = body['choices']
188
+ unless choices.is_a?(Array) && choices.first.is_a?(Hash)
189
+ raise Error, 'Kimi answered without a choice'
190
+ end
191
+
192
+ choice = choices.first
193
+ message = choice['message']
194
+ raise Error, 'Kimi answered without a message' unless message.is_a?(Hash)
195
+ raise Error, "Kimi did not finish: #{choice['finish_reason']}" unless choice['finish_reason'] == 'stop'
196
+
197
+ text = message['content'].to_s.strip
198
+ raise Error, 'Kimi answered with no content' if text.empty?
199
+
200
+ text
201
+ end
202
+
203
+ # The service's own explanation where it gave one, the status line
204
+ # otherwise. Neither carries the token, and the settings are never logged
205
+ # or raised wholesale.
206
+ def reason(response)
207
+ body = JSON.parse(response.body.to_s)
208
+ error = body['error']
209
+ return error['message'].to_s if error.is_a?(Hash) && !error['message'].to_s.empty?
210
+
211
+ response.message.to_s
212
+ rescue JSON::ParserError
213
+ response.message.to_s
214
+ end
215
+ end
216
+ end