d5cabfe5c6
Compact Language Detector v3 (CLD3) is the successor of CLD2, which was used in the previous implementation. CLD3 includes improvements since CLD2, and supports newer compilers. On the other hand, it has additional requirements and cld3-ruby, the FFI of CLD3 for Ruby, is still new and may be still inmature. Though CLD3 is named after CLD2, it is implemented with a neural network model, different from the old implementation, which is based on a Naïve Bayesian classifier. CLD3 supports newer compilers, such as GCC 6. CLD2 is not compatible with GCC 6 because it assigns negative values to varibales typed unsigned. (see internal/cld_generated_cjk_uni_prop_80.cc) The support for GCC 6 and newer compilers are essential today, when some server operating system such as Ubuntu Server 16.10 has GCC 6 by default. On the one hand, CLD3 requires C++11 support. Environments with old compilers such as Ubuntu Server 14.04 needs to update the system or install a newer compiler. CLD3 needs protocol buffers as a new dependency. However,it is not considered problematic because major server operating systems, CentOS and Ubuntu Server provide them. The FFI cld3-ruby was written by me (Akihiko Odaki) for use in Mastodon. It is still new and may be inmature, but confirmed to pass existing tests.
87 lines
2.4 KiB
Ruby
87 lines
2.4 KiB
Ruby
# frozen_string_literal: true
|
|
require 'rails_helper'
|
|
|
|
describe LanguageDetector do
|
|
describe 'to_iso_s' do
|
|
it 'detects english language for basic strings' do
|
|
strings = [
|
|
"Hello and welcome to mastodon",
|
|
"I'd rather not!",
|
|
"a lot of people just want to feel righteous all the time and that's all that matters",
|
|
]
|
|
strings.each do |string|
|
|
result = described_class.new(string).to_iso_s
|
|
|
|
expect(result).to eq(:en), string
|
|
end
|
|
end
|
|
|
|
it 'detects spanish language' do
|
|
string = 'Obtener un Hola y bienvenidos a Mastodon'
|
|
result = described_class.new(string).to_iso_s
|
|
|
|
expect(result).to eq :es
|
|
end
|
|
|
|
describe 'when language can\'t be detected' do
|
|
it 'uses default locale when sent an empty document' do
|
|
result = described_class.new('').to_iso_s
|
|
expect(result).to eq :en
|
|
end
|
|
|
|
describe 'because of a URL' do
|
|
it 'uses default locale when sent just a URL' do
|
|
string = 'http://example.com/media/2kFTgOJLXhQf0g2nKB4'
|
|
cld_result = CLD3::NNetLanguageIdentifier.new(0, 2048).find_language(string)
|
|
expect(cld_result).not_to eq :en
|
|
|
|
result = described_class.new(string).to_iso_s
|
|
|
|
expect(result).to eq :en
|
|
end
|
|
end
|
|
|
|
describe 'with an account' do
|
|
it 'uses the account locale when present' do
|
|
account = double(user_locale: 'fr')
|
|
result = described_class.new('', account).to_iso_s
|
|
|
|
expect(result).to eq :fr
|
|
end
|
|
|
|
it 'uses default locale when account is present but has no locale' do
|
|
account = double(user_locale: nil)
|
|
result = described_class.new('', account).to_iso_s
|
|
|
|
expect(result).to eq :en
|
|
end
|
|
end
|
|
|
|
describe 'with an `en` default locale' do
|
|
it 'uses the default locale' do
|
|
string = ''
|
|
result = described_class.new(string).to_iso_s
|
|
|
|
expect(result).to eq :en
|
|
end
|
|
end
|
|
|
|
describe 'with a non-`en` default locale' do
|
|
around(:each) do |example|
|
|
before = I18n.default_locale
|
|
I18n.default_locale = :ja
|
|
example.run
|
|
I18n.default_locale = before
|
|
end
|
|
|
|
it 'uses the default locale' do
|
|
string = ''
|
|
result = described_class.new(string).to_iso_s
|
|
|
|
expect(result).to eq :ja
|
|
end
|
|
end
|
|
end
|
|
end
|
|
end
|