Initial commit
This commit is contained in:
@@ -0,0 +1,36 @@
|
||||
require_relative './memory'
|
||||
require_relative './redis'
|
||||
|
||||
module DataRepository
|
||||
module Backend
|
||||
class Detector
|
||||
def initialize(backend_or_connection=nil)
|
||||
@backend_or_connection = backend_or_connection
|
||||
end #initialize
|
||||
|
||||
def detect
|
||||
# TYPE CHECKING OMG!! SEND THE CRAFTSMANSHIPTROOPERS!!
|
||||
return backend_or_connection if is_backend?
|
||||
return Backend::Redis.new(backend_or_connection) if is_redis?
|
||||
return default_backend
|
||||
end #detect
|
||||
|
||||
private
|
||||
|
||||
attr_reader :backend_or_connection
|
||||
|
||||
def default_backend
|
||||
Backend::Memory.new
|
||||
end #default_backend
|
||||
|
||||
def is_backend?
|
||||
!!(backend_or_connection.class.name =~ /^DataRepository::Backend/)
|
||||
end #is_backend?
|
||||
|
||||
def is_redis?
|
||||
backend_or_connection.respond_to? :zremrangebyscore
|
||||
end #is_redis?
|
||||
end # Detector
|
||||
end # Backend
|
||||
end # DataRepository
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
require 'set'
|
||||
require 'json'
|
||||
|
||||
module DataRepository
|
||||
module Backend
|
||||
class Memory < Hash
|
||||
def store(key, data, options={})
|
||||
data = data.to_a if data.is_a?(Set) # OMG FIXME
|
||||
super(key, JSON.parse(data.to_json))
|
||||
end
|
||||
|
||||
def fetch(key, options={})
|
||||
super key
|
||||
end
|
||||
|
||||
def exists?(key)
|
||||
self.has_key?(key)
|
||||
end
|
||||
|
||||
# Not supported, so just call data
|
||||
def transaction(&block)
|
||||
block.call
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
require 'redis'
|
||||
require_relative 'redis/set'
|
||||
require_relative 'redis/string'
|
||||
|
||||
module DataRepository
|
||||
module Backend
|
||||
class Redis
|
||||
|
||||
HANDLERS = {
|
||||
'set' => Backend::Redis::Set,
|
||||
'string' => Backend::Redis::String
|
||||
}
|
||||
|
||||
def initialize(redis=::Redis.new)
|
||||
@redis = redis
|
||||
end #initialize
|
||||
|
||||
def store(key, data, options={})
|
||||
persister_for(data).new(redis).store(key, data)
|
||||
expire_in(options.fetch(:expiration, nil), key)
|
||||
end #store
|
||||
|
||||
def fetch(key)
|
||||
return nil unless redis.exists(key)
|
||||
retriever_for(key).new(redis).fetch(key)
|
||||
end #fetch
|
||||
|
||||
def keys
|
||||
redis.keys
|
||||
end #keys
|
||||
|
||||
def exists?(key)
|
||||
redis.exists(key)
|
||||
end #exists?
|
||||
|
||||
def delete(key)
|
||||
redis.del(key)
|
||||
end #delete
|
||||
|
||||
# Not supported, so just call data
|
||||
def transaction(&block)
|
||||
block.call
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
attr_reader :redis
|
||||
|
||||
def expire_in(seconds, key)
|
||||
!!seconds && redis.expire(key, seconds)
|
||||
end #expire_in
|
||||
|
||||
def retriever_for(key)
|
||||
HANDLERS.fetch(redis.type(key), Backend::Redis::String)
|
||||
end #retriever_for
|
||||
|
||||
def persister_for(data)
|
||||
HANDLERS.fetch(data.class.to_s.downcase, Backend::Redis::String)
|
||||
end #persister_for(data)
|
||||
end # Redis
|
||||
end # Backend
|
||||
end # DataRepository
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
require 'redis'
|
||||
|
||||
module DataRepository
|
||||
module Backend
|
||||
class Redis
|
||||
class Set
|
||||
def initialize(redis=Redis.new)
|
||||
@redis = redis
|
||||
end #initialize
|
||||
|
||||
def store(key, data)
|
||||
workaround_until_resque_supports_latest_redis_gem(key, data)
|
||||
end #store
|
||||
|
||||
def fetch(key)
|
||||
redis.smembers key
|
||||
end #fetch
|
||||
|
||||
private
|
||||
|
||||
attr_reader :redis
|
||||
|
||||
def workaround_until_resque_supports_latest_redis_gem(key, data)
|
||||
redis.multi do
|
||||
data.to_a.each { |item| redis.sadd(key, item) }
|
||||
end
|
||||
end #workaround_until_resque_supports_latest_redis_gem
|
||||
end # Set
|
||||
end # Redis
|
||||
end # Backend
|
||||
end # DataRepository
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
require 'json'
|
||||
require 'redis'
|
||||
|
||||
module DataRepository
|
||||
module Backend
|
||||
class Redis
|
||||
class String
|
||||
def initialize(redis=Redis.new)
|
||||
@redis = redis
|
||||
end #initialize
|
||||
|
||||
def store(key, data)
|
||||
redis.set key, data.to_json
|
||||
end #store
|
||||
|
||||
def fetch(key)
|
||||
JSON.parse redis.get(key)
|
||||
end #fetch
|
||||
|
||||
private
|
||||
|
||||
attr_reader :redis
|
||||
end # String
|
||||
end # Redis
|
||||
end # Backend
|
||||
end # DataRepository
|
||||
|
||||
@@ -0,0 +1,135 @@
|
||||
require 'sequel'
|
||||
require 'uuidtools'
|
||||
|
||||
module DataRepository
|
||||
module Backend
|
||||
class Sequel
|
||||
PAGE = 1
|
||||
PER_PAGE = 300
|
||||
ARRAY_RE = %r{\[.*\]}
|
||||
|
||||
def initialize(db, relation=nil)
|
||||
@db = db
|
||||
@relation = relation.to_sym
|
||||
::Sequel.extension(:pagination)
|
||||
::Sequel.extension(:connection_validator)
|
||||
@db.extension :pg_array if postgres?(@db)
|
||||
end
|
||||
|
||||
def collection(filters={}, available_filters=[])
|
||||
apply_filters(db[relation], filters, available_filters)
|
||||
end
|
||||
|
||||
def store(key, data={})
|
||||
naive_upsert_exposed_to_race_conditions(data)
|
||||
end
|
||||
|
||||
def fetch(value, key=nil)
|
||||
if key.nil?
|
||||
parse( db[relation].where(id: value).first )
|
||||
else
|
||||
parse( db[relation].where(key.to_sym => value).first )
|
||||
end
|
||||
end
|
||||
|
||||
def delete(key)
|
||||
db[relation].where(id: key).delete
|
||||
end
|
||||
|
||||
def next_id
|
||||
UUIDTools::UUID.timestamp_create
|
||||
end
|
||||
|
||||
def apply_filters(dataset, filters={}, available_filters=[])
|
||||
return dataset if filters.nil? || filters.empty?
|
||||
available_filters = symbolize_elements(available_filters)
|
||||
|
||||
filters = symbolize_keys(filters).select { |key, value|
|
||||
available_filters.include?(key)
|
||||
} unless available_filters.empty?
|
||||
|
||||
dataset.where(filters)
|
||||
end
|
||||
|
||||
def paginate(dataset, filter={}, record_count=nil)
|
||||
page, per_page = pagination_params_from(filter)
|
||||
dataset.paginate(page, per_page, record_count)
|
||||
end
|
||||
|
||||
def transaction(&block)
|
||||
db.transaction(&block)
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
attr_reader :relation, :db
|
||||
|
||||
def naive_upsert_exposed_to_race_conditions(data={})
|
||||
data = send("serialize_for_#{backend_type}", data)
|
||||
insert(data) unless update(data)
|
||||
end
|
||||
|
||||
def insert(data={})
|
||||
db[relation].insert(data)
|
||||
end
|
||||
|
||||
def update(data={})
|
||||
db[relation].where(id: data.fetch(:id)).update(data) != 0
|
||||
end
|
||||
|
||||
def serialize_for_postgres(data)
|
||||
Hash[
|
||||
data.map { |key, value|
|
||||
value = ::Sequel.pg_array(value) if value.is_a?(Array) && !value.empty?
|
||||
[key, value]
|
||||
}
|
||||
]
|
||||
end
|
||||
|
||||
def serialize_for_other_database(data={})
|
||||
Hash[
|
||||
data.map { |key, value|
|
||||
value = value.to_s if value.is_a?(Array)
|
||||
[key, value]
|
||||
}
|
||||
]
|
||||
end
|
||||
|
||||
def backend_type
|
||||
postgres?(db) ? :postgres : :other_database
|
||||
end
|
||||
|
||||
def postgres?(db)
|
||||
db.database_type == :postgres
|
||||
end
|
||||
|
||||
def parse(attributes={})
|
||||
return unless attributes
|
||||
return attributes if postgres?(db)
|
||||
|
||||
Hash[
|
||||
attributes.map do |key, value|
|
||||
value = JSON.parse(value) if value =~ ARRAY_RE
|
||||
[key, value]
|
||||
end
|
||||
]
|
||||
end
|
||||
|
||||
def symbolize_elements(array=[])
|
||||
array.map { |k| k.to_sym}
|
||||
end
|
||||
|
||||
def symbolize_keys(hash={})
|
||||
Hash[ hash.map { |k, v| [k.to_sym, v] } ]
|
||||
end
|
||||
|
||||
def pagination_params_from(filter)
|
||||
page = (filter.delete(:page) || PAGE).to_i
|
||||
per_page = (filter.delete(:per_page) || PER_PAGE).to_i
|
||||
|
||||
[page, per_page]
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
require 'fileutils'
|
||||
|
||||
module DataRepository
|
||||
module Filesystem
|
||||
class Local
|
||||
DEFAULT_PREFIX = File.join(File.dirname(__FILE__), '..', 'tmp')
|
||||
|
||||
def initialize(base_directory=DEFAULT_PREFIX)
|
||||
@base_directory = base_directory
|
||||
end
|
||||
|
||||
def create_base_directory
|
||||
FileUtils.mkpath @base_directory unless exists? @base_directory
|
||||
end
|
||||
|
||||
def store(path, data)
|
||||
FileUtils.mkpath(File.dirname(fullpath_for(path)))
|
||||
|
||||
File.open(fullpath_for(path), 'wb') do |file|
|
||||
data.rewind if data.eof?
|
||||
if data.respond_to?(:bucket)
|
||||
data.read { |chunk| file.write(chunk) }
|
||||
else
|
||||
chunk = data.gets
|
||||
while chunk
|
||||
file.write(chunk)
|
||||
chunk = data.gets
|
||||
end
|
||||
end
|
||||
end
|
||||
path
|
||||
end
|
||||
|
||||
def fetch(path)
|
||||
File.open(fullpath_for(path), 'r')
|
||||
end
|
||||
|
||||
def exists?(path)
|
||||
File.exists?(fullpath_for(path))
|
||||
end
|
||||
|
||||
# Use from controlled environments always
|
||||
def remove(path)
|
||||
if exists?(path)
|
||||
File.delete(fullpath_for(path))
|
||||
end
|
||||
end
|
||||
|
||||
def fullpath_for(path)
|
||||
File.join(base_directory, path)
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
attr_reader :base_directory
|
||||
|
||||
def targets_for(path)
|
||||
fullpath = fullpath_for(path)
|
||||
[
|
||||
Dir.glob(fullpath),
|
||||
Dir.glob("#{fullpath}/*"),
|
||||
Dir.glob("#{fullpath}/**/*")
|
||||
].flatten
|
||||
.uniq
|
||||
.delete_if { |entry| dot_directory?(entry) }
|
||||
end
|
||||
|
||||
def dot_directory?(path)
|
||||
path == '.' || path == '..'
|
||||
end
|
||||
|
||||
def relative_path_for(path, base_directory)
|
||||
(path.split('/') - base_directory.split('/')).join('/')
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
require 'uuidtools'
|
||||
require_relative 'backend/detector'
|
||||
require_relative 'backend/memory'
|
||||
|
||||
module DataRepository
|
||||
def self.new(backend_or_connection=nil)
|
||||
backend = Backend::Detector.new(backend_or_connection).detect
|
||||
Repository.new(backend)
|
||||
end # DataRepository.new
|
||||
|
||||
class Repository
|
||||
def initialize(storage=Backend::Memory.new)
|
||||
@storage = storage
|
||||
end
|
||||
|
||||
def backend
|
||||
@storage
|
||||
end
|
||||
|
||||
def store(key, data, options={})
|
||||
storage.store(key.to_s, data, options)
|
||||
end
|
||||
|
||||
def fetch(key)
|
||||
storage.fetch(key.to_s)
|
||||
end
|
||||
|
||||
def delete(key)
|
||||
storage.delete(key.to_s)
|
||||
end
|
||||
|
||||
def exists?(key)
|
||||
storage.exists?(key)
|
||||
end
|
||||
|
||||
def keys
|
||||
storage.keys
|
||||
end
|
||||
|
||||
def next_id
|
||||
UUIDTools::UUID.timestamp_create
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
attr_reader :storage
|
||||
end # Handler
|
||||
end # DataRepository
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
require 'minitest/autorun'
|
||||
require 'redis'
|
||||
require_relative '../spec_helper'
|
||||
require_relative '../../repository'
|
||||
|
||||
describe DataRepository do
|
||||
describe 'DataRepository.new' do
|
||||
it 'instantiates a repository with a memory backend' do
|
||||
repository = DataRepository.new
|
||||
repository.must_be_instance_of DataRepository::Repository
|
||||
repository.backend.must_be_instance_of DataRepository::Backend::Memory
|
||||
end
|
||||
|
||||
it 'detects the backend if a repository or DB connection is passed' do
|
||||
repository = DataRepository.new(Redis.new)
|
||||
repository.must_be_instance_of DataRepository::Repository
|
||||
repository.backend.must_be_instance_of DataRepository::Backend::Redis
|
||||
end
|
||||
end # DataRepository.new
|
||||
end # DataRepository
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
require 'minitest/autorun'
|
||||
require 'redis'
|
||||
require_relative '../../spec_helper'
|
||||
require_relative '../../../backend/detector'
|
||||
|
||||
include DataRepository
|
||||
|
||||
describe Backend::Detector do
|
||||
describe '#detect' do
|
||||
it 'returns a memory backend if no connection or backend passed' do
|
||||
Backend::Detector.new.detect.must_be_instance_of Backend::Memory
|
||||
end
|
||||
|
||||
it "returns the passed object if it's a backend" do
|
||||
memory_backend = Backend::Memory.new
|
||||
detector = Backend::Detector.new(memory_backend)
|
||||
detector.detect.must_be_instance_of Backend::Memory
|
||||
|
||||
redis_backend = Backend::Redis.new
|
||||
detector = Backend::Detector.new(redis_backend)
|
||||
detector.detect.must_be_instance_of Backend::Redis
|
||||
end
|
||||
|
||||
it 'returns a redis backend if a redis connection is passed' do
|
||||
detector = Backend::Detector.new(Redis.new)
|
||||
detector.detect.must_be_instance_of Backend::Redis
|
||||
end
|
||||
end #detect
|
||||
end # Backend::Detector
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
require 'minitest/autorun'
|
||||
require 'ostruct'
|
||||
require_relative '../../spec_helper'
|
||||
require_relative '../../../repository'
|
||||
require_relative '../../../backend/memory'
|
||||
|
||||
include DataRepository
|
||||
|
||||
describe Repository do
|
||||
before do
|
||||
@repository = Repository.new(Backend::Memory.new)
|
||||
end
|
||||
|
||||
describe '#store' do
|
||||
it 'persists a data structure in the passed key' do
|
||||
data = { id: 5 }
|
||||
key = data.fetch(:id)
|
||||
|
||||
@repository.keys.wont_include key.to_s
|
||||
@repository.store(key, data)
|
||||
@repository.keys.must_include key.to_s
|
||||
end
|
||||
|
||||
it 'stringifies symbols in the persisted data structure' do
|
||||
data = { id: 5 }
|
||||
key = data.fetch(:id)
|
||||
|
||||
@repository.store(key, data)
|
||||
retrieved_data = @repository.fetch(key)
|
||||
|
||||
retrieved_data.keys.wont_include :id
|
||||
retrieved_data.keys.must_include 'id'
|
||||
end
|
||||
end #store
|
||||
|
||||
describe '#fetch' do
|
||||
it 'retrieves a data structure from a key' do
|
||||
data = { id: 5 }
|
||||
key = data.fetch(:id)
|
||||
|
||||
@repository.store(key, data)
|
||||
retrieved_data = @repository.fetch(key.to_s)
|
||||
retrieved_data.fetch('id').must_equal data.fetch(:id)
|
||||
end
|
||||
end #fetch
|
||||
|
||||
describe '#delete' do
|
||||
it 'deletes a key' do
|
||||
data = { id: 5 }
|
||||
key = data.fetch(:id)
|
||||
|
||||
@repository.store(key, data)
|
||||
@repository.fetch(key.to_s).wont_be_nil
|
||||
|
||||
@repository.delete(key)
|
||||
lambda { @repository.fetch(key.to_s) }.must_raise KeyError
|
||||
end
|
||||
end #delete
|
||||
|
||||
describe '#keys' do
|
||||
it 'returns all stored keys, stringified' do
|
||||
data = { id: 5 }
|
||||
key = data.fetch(:id)
|
||||
|
||||
@repository.store(key, data)
|
||||
@repository.keys.must_equal [key.to_s]
|
||||
end
|
||||
end #keys
|
||||
|
||||
describe '#exists?' do
|
||||
it 'returns if key exists' do
|
||||
data = { id: 5 }
|
||||
key = data.fetch(:id)
|
||||
|
||||
@repository.exists?(key.to_s).must_equal false
|
||||
@repository.store(key, data)
|
||||
@repository.exists?(key.to_s).must_equal true
|
||||
end
|
||||
end #exists?
|
||||
end # Repository
|
||||
|
||||
@@ -0,0 +1,111 @@
|
||||
require 'minitest/autorun'
|
||||
require 'set'
|
||||
require_relative '../../spec_helper'
|
||||
require_relative '../../../backend/redis'
|
||||
require_relative '../../../repository'
|
||||
|
||||
include DataRepository
|
||||
|
||||
describe Backend::Redis do
|
||||
before do
|
||||
@connection = Redis.new
|
||||
@connection.select 8
|
||||
@connection.flushdb
|
||||
|
||||
storage = Backend::Redis.new(@connection)
|
||||
@repository = Repository.new(storage)
|
||||
end
|
||||
|
||||
describe '#store' do
|
||||
it 'persists a data structure in the passed key' do
|
||||
data = { id: 5 }
|
||||
key = data.fetch(:id)
|
||||
|
||||
@repository.keys.wont_include key.to_s
|
||||
@repository.store(key, data)
|
||||
@repository.keys.must_include key.to_s
|
||||
|
||||
set = Set.new
|
||||
set.add 1
|
||||
set.add 2
|
||||
|
||||
@repository.store('bogus_key', set)
|
||||
@repository.fetch('bogus_key').must_be_kind_of Enumerable
|
||||
end
|
||||
|
||||
it 'stringifies symbols in the persisted data structe' do
|
||||
data = { id: 5 }
|
||||
key = data.fetch(:id)
|
||||
|
||||
@repository.store(key, data)
|
||||
retrieved_data = @repository.fetch(key)
|
||||
|
||||
retrieved_data.keys.wont_include :id
|
||||
retrieved_data.keys.must_include 'id'
|
||||
end
|
||||
|
||||
it 'sets key expiration in seconds if expiration option passed' do
|
||||
data = { id: 5 }
|
||||
key = data.fetch(:id)
|
||||
expiration = 1
|
||||
|
||||
@repository.store(key, data, expiration: expiration)
|
||||
retrieved_data = @repository.fetch(key)
|
||||
retrieved_data.keys.must_include 'id'
|
||||
@connection.get(key).wont_be_nil
|
||||
sleep(expiration.to_f + 1.0 / 1.0)
|
||||
@connection.get(key).must_be_nil
|
||||
end
|
||||
end #store
|
||||
|
||||
describe '#fetch' do
|
||||
it 'retrieves a data structure from a key' do
|
||||
data = { id: 5 }
|
||||
key = data.fetch(:id)
|
||||
|
||||
@repository.store(key, data)
|
||||
retrieved_data = @repository.fetch(key)
|
||||
|
||||
retrieved_data.fetch('id').must_equal data.fetch(:id)
|
||||
end
|
||||
|
||||
it 'returns nil if key does not exist' do
|
||||
@repository.fetch('non_existent_key').must_equal nil
|
||||
end
|
||||
end #fetch
|
||||
|
||||
describe '#delete' do
|
||||
it 'deletes a key' do
|
||||
data = { id: 5 }
|
||||
key = data.fetch(:id)
|
||||
|
||||
@repository.store(key, data)
|
||||
@repository.fetch(key).wont_be_nil
|
||||
|
||||
@repository.delete(key)
|
||||
@repository.fetch(key).must_be_nil
|
||||
end
|
||||
end #delete
|
||||
|
||||
describe '#keys' do
|
||||
it 'returns all stored keys, stringified' do
|
||||
data = { id: 5 }
|
||||
key = data.fetch(:id)
|
||||
|
||||
@repository.store(key, data)
|
||||
@repository.keys.must_equal [key.to_s]
|
||||
end
|
||||
end #keys
|
||||
|
||||
describe '#exists?' do
|
||||
it 'returns if key exists' do
|
||||
data = { id: 5 }
|
||||
key = data.fetch(:id)
|
||||
|
||||
@repository.exists?(key.to_s).must_equal false
|
||||
@repository.store(key, data)
|
||||
@repository.exists?(key.to_s).must_equal true
|
||||
end
|
||||
end #exists?
|
||||
end # Backend::Redis
|
||||
|
||||
@@ -0,0 +1,104 @@
|
||||
require_relative '../../../backend/sequel'
|
||||
require_relative '../../../../../app/models/visualization/member'
|
||||
|
||||
include CartoDB
|
||||
|
||||
describe DataRepository::Backend::Sequel do
|
||||
before do
|
||||
db = SequelRails.connection
|
||||
db.create_table :visualizations do
|
||||
UUID :id, primary_key: true
|
||||
String :name
|
||||
String :display_name
|
||||
String :title
|
||||
String :description
|
||||
String :license
|
||||
String :source
|
||||
String :tags
|
||||
String :map_id
|
||||
String :active_layer_id
|
||||
String :type
|
||||
String :privacy
|
||||
String :encrypted_password
|
||||
String :password_salt
|
||||
UUID :permission_id
|
||||
Boolean :locked
|
||||
String :parent_id
|
||||
String :kind
|
||||
String :prev_id
|
||||
String :next_id
|
||||
String :slide_transition_options
|
||||
String :active_child
|
||||
end
|
||||
|
||||
db.create_table :overlays do
|
||||
String :id, null: false, primary_key: true
|
||||
Integer :order, null: false
|
||||
String :options, text: true
|
||||
String :type
|
||||
String :visualization_id, index: true
|
||||
end
|
||||
|
||||
Visualization.repository = DataRepository::Backend::Sequel.new(db, :visualizations)
|
||||
Overlay.repository = DataRepository::Backend::Sequel.new(db, :overlays)
|
||||
end
|
||||
|
||||
describe '#store' do
|
||||
it 'inserts a visualization' do
|
||||
member = Visualization::Member.new(
|
||||
name: 'visualization 1',
|
||||
tags: ['foo', 'bar']
|
||||
)
|
||||
member.store
|
||||
|
||||
rehydrated_member = Visualization::Member.new(id: member.id)
|
||||
rehydrated_member.fetch
|
||||
rehydrated_member.name.should eq member.name
|
||||
end
|
||||
|
||||
it 'updates the visualization if existing' do
|
||||
member = Visualization::Member.new(
|
||||
name: 'visualization 1',
|
||||
tags: ['foo', 'bar']
|
||||
)
|
||||
member.store
|
||||
Visualization.repository.collection(id: member.id).to_a.size.should eq 1
|
||||
|
||||
member.store
|
||||
Visualization.repository.collection(id: member.id).to_a.size.should eq 1
|
||||
|
||||
Visualization::Member.new(id: member.id).fetch.store
|
||||
Visualization.repository.collection(id: member.id).to_a.size.should eq 1
|
||||
end
|
||||
end
|
||||
|
||||
describe '#delete' do
|
||||
it 'deletes a visualization from persistence' do
|
||||
member = Visualization::Member.new(
|
||||
name: 'visualization 1',
|
||||
tags: ['foo', 'bar']
|
||||
).store
|
||||
|
||||
id = member.id
|
||||
Visualization.repository.fetch(id).nil?.should eq false
|
||||
member.delete
|
||||
Visualization.repository.fetch(id).nil?.should eq true
|
||||
end
|
||||
end
|
||||
|
||||
describe '#collection' do
|
||||
it 'gets a collection of records using the passed filter' do
|
||||
Visualization::Member.new(
|
||||
name: 'visualization 1',
|
||||
map_id: 1
|
||||
).store
|
||||
Visualization::Member.new(
|
||||
name: 'visualization 2',
|
||||
map_id: 1
|
||||
).store
|
||||
|
||||
records = Visualization.repository.collection(map_id: 1)
|
||||
records.to_a.size.should eq 2
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,53 @@
|
||||
require 'minitest/autorun'
|
||||
require 'stringio'
|
||||
require_relative '../../../filesystem/local'
|
||||
|
||||
include DataRepository::Filesystem
|
||||
|
||||
describe Local do
|
||||
before do
|
||||
@data = StringIO.new(Time.now.to_f.to_s)
|
||||
@path = File.join( (0..2).map { Time.now.to_f.to_s } )
|
||||
@prefix = File.join(Local::DEFAULT_PREFIX, Time.now.to_i.to_s)
|
||||
end
|
||||
|
||||
after do
|
||||
@data.close
|
||||
FileUtils.rmtree(@prefix)
|
||||
FileUtils.rmtree(Local::DEFAULT_PREFIX)
|
||||
end
|
||||
|
||||
describe '#initialize' do
|
||||
it 'sets the storage prefix to Local::DEFAULT_PREFIX by default' do
|
||||
File.exists?( File.join(Local::DEFAULT_PREFIX, @path) ).must_equal false
|
||||
|
||||
path = Local.new.store(@path, @data)
|
||||
File.exists?( File.join(Local::DEFAULT_PREFIX, @path) ).must_equal true
|
||||
end
|
||||
end #initialize
|
||||
|
||||
describe '#store' do
|
||||
it 'stores data in the specified path' do
|
||||
filesystem = Local.new(@prefix)
|
||||
path = filesystem.store(@path, @data)
|
||||
|
||||
@data.rewind
|
||||
|
||||
stored_data = File.open( File.join(@prefix, path) )
|
||||
stored_data.read.must_equal @data.read
|
||||
end
|
||||
end #store
|
||||
|
||||
describe '#fetch' do
|
||||
it 'retrieves data from the specified path' do
|
||||
filesystem = Local.new(@prefix)
|
||||
path = filesystem.store(@path, @data)
|
||||
|
||||
@data.rewind
|
||||
|
||||
stored_data = Local.new(@prefix).fetch(path)
|
||||
stored_data.read.must_equal @data.read
|
||||
end
|
||||
end #fetch
|
||||
end # Local
|
||||
|
||||
@@ -0,0 +1,123 @@
|
||||
require 'minitest/autorun'
|
||||
require_relative '../../../structures/collection'
|
||||
|
||||
include DataRepository
|
||||
|
||||
describe Collection do
|
||||
before do
|
||||
@repository = DataRepository::Repository.new
|
||||
@dummy_class = Class.new do
|
||||
attr_accessor :id
|
||||
def initialize(arguments={}); self.id = arguments.fetch(:id); end
|
||||
def fetch; self; end
|
||||
def to_hash; { id: id }; end
|
||||
def ==(other); id.to_s == other.id.to_s; end
|
||||
end
|
||||
|
||||
@defaults = { repository: @repository, member_class: @dummy_class}
|
||||
end
|
||||
|
||||
describe '#add' do
|
||||
it 'adds a member to the collection' do
|
||||
member = @dummy_class.new(id: 1)
|
||||
collection = Collection.new(@defaults)
|
||||
collection.add(member)
|
||||
collection.to_a.must_include member
|
||||
end
|
||||
end
|
||||
|
||||
describe '#delete' do
|
||||
it 'deletes a member from the collection' do
|
||||
member = @dummy_class.new(id: 1)
|
||||
collection = Collection.new(@defaults)
|
||||
collection.add(member)
|
||||
collection.delete(member)
|
||||
collection.to_a.wont_include member
|
||||
end
|
||||
end #delete
|
||||
|
||||
describe '#each' do
|
||||
it 'yields members of the collection as the initialized member_class' do
|
||||
member = @dummy_class.new(id: 1)
|
||||
collection = Collection.new(@defaults)
|
||||
collection.add(member)
|
||||
collection.store
|
||||
|
||||
rehydrated_collection =
|
||||
Collection.new(@defaults.merge(signature: collection.signature))
|
||||
rehydrated_collection.fetch
|
||||
rehydrated_collection.to_a.first.must_be_instance_of @dummy_class
|
||||
end
|
||||
|
||||
it 'returns an enumerator if no block given' do
|
||||
member = @dummy_class.new(id: 1)
|
||||
collection = Collection.new({ repository: @repository })
|
||||
collection.add(member)
|
||||
collection.store
|
||||
|
||||
rehydrated_collection =
|
||||
Collection.new(@defaults.merge(signature: collection.signature))
|
||||
rehydrated_collection.fetch
|
||||
|
||||
enumerator = rehydrated_collection.each
|
||||
enumerator.next.must_be_instance_of @dummy_class
|
||||
end
|
||||
end #each
|
||||
|
||||
describe '#fetch' do
|
||||
it 'resets the collection with data from the data repository' do
|
||||
member1 = @dummy_class.new(id: 1)
|
||||
member2 = @dummy_class.new(id: 2)
|
||||
collection = Collection.new(@defaults)
|
||||
collection.add(member1)
|
||||
collection.store
|
||||
|
||||
rehydrated_collection =
|
||||
Collection.new(@defaults.merge(signature: collection.signature))
|
||||
rehydrated_collection.add(member2)
|
||||
|
||||
rehydrated_collection.to_a.must_include(member2)
|
||||
rehydrated_collection.to_a.wont_include(member1)
|
||||
rehydrated_collection.fetch
|
||||
rehydrated_collection.to_a.must_include(member1)
|
||||
rehydrated_collection.to_a.wont_include(member2)
|
||||
end
|
||||
|
||||
it 'empties the collection if it was not persisted to the repository' do
|
||||
member = @dummy_class.new(id: 1)
|
||||
collection = Collection.new(@defaults)
|
||||
collection.add(member)
|
||||
collection.to_a.length.must_equal 1
|
||||
collection.fetch
|
||||
collection.to_a.must_be_empty
|
||||
end
|
||||
end #fetch
|
||||
|
||||
describe '#store' do
|
||||
it 'persists the collection to the data repository' do
|
||||
member = @dummy_class.new(id: 1)
|
||||
collection = Collection.new(@defaults)
|
||||
collection.add(member)
|
||||
collection.store
|
||||
|
||||
rehydrated_collection =
|
||||
Collection.new(@defaults.merge(signature: collection.signature))
|
||||
rehydrated_collection.fetch
|
||||
rehydrated_collection.map { |member| member.id }.must_include member.id
|
||||
end
|
||||
end #store
|
||||
|
||||
describe '#to_json' do
|
||||
it 'renders a JSON representation of the collection' do
|
||||
member = @dummy_class.new(id: 1)
|
||||
collection = Collection.new(@defaults)
|
||||
collection.add(member)
|
||||
collection.store
|
||||
|
||||
representation = JSON.parse(collection.to_json)
|
||||
representation.size.must_equal 1
|
||||
representation.first.fetch('id').must_equal member.id
|
||||
end
|
||||
end #to_json
|
||||
end # Collection
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
require 'ostruct'
|
||||
require 'set'
|
||||
require_relative '../repository'
|
||||
|
||||
module DataRepository
|
||||
class Collection
|
||||
include Enumerable
|
||||
|
||||
INTERFACE = %w{ signature add delete store fetch each to_json repository count } + Enumerable.instance_methods
|
||||
|
||||
attr_reader :signature
|
||||
attr_accessor :storage
|
||||
|
||||
def initialize(arguments={})
|
||||
@storage = Set.new
|
||||
@member_class = arguments.fetch(:member_class, OpenStruct)
|
||||
@repository = arguments.fetch(:repository, Repository.new)
|
||||
@signature = arguments.fetch(:signature, @repository.next_id)
|
||||
end #initialize
|
||||
|
||||
def add(member)
|
||||
storage.add(member)
|
||||
self
|
||||
end #add
|
||||
|
||||
def delete(member)
|
||||
storage.delete(member)
|
||||
self
|
||||
end #delete
|
||||
|
||||
def delete_if(&block)
|
||||
storage.delete_if(&block)
|
||||
self
|
||||
end
|
||||
|
||||
def each(&block)
|
||||
return storage.each(&block)
|
||||
Enumerator.new(self, :each)
|
||||
end #each
|
||||
|
||||
def fetch
|
||||
self.storage = Set[*repository.fetch(signature)].map do |attributes|
|
||||
puts attributes.inspect
|
||||
member_class.new(attributes)
|
||||
end
|
||||
|
||||
self
|
||||
rescue => exception
|
||||
storage.clear
|
||||
self
|
||||
end #fetch
|
||||
|
||||
def store
|
||||
repository.store(signature, storage.map(&:id).to_a)
|
||||
self
|
||||
end #store
|
||||
|
||||
def to_json(*args)
|
||||
map { |member| member.to_hash }.to_json(*args)
|
||||
end #to_json
|
||||
|
||||
def count
|
||||
storage.count
|
||||
end #count
|
||||
|
||||
private
|
||||
|
||||
attr_reader :repository, :member_class
|
||||
|
||||
def members
|
||||
storage.each { |member_id| yield member_class.new(id: member_id).fetch }
|
||||
end #members
|
||||
end # Collection
|
||||
end # DataRepository
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
require 'active_support/time'
|
||||
require_relative 'service_usage_metrics'
|
||||
|
||||
module CartoDB
|
||||
# The purpose of this class is to encapsulate storage of usage metrics.
|
||||
# This shall be used for billing, quota checking and metrics.
|
||||
class GeocoderUsageMetrics < ServiceUsageMetrics
|
||||
|
||||
VALID_METRICS = [
|
||||
:total_requests,
|
||||
:failed_responses,
|
||||
:success_responses,
|
||||
:empty_responses,
|
||||
].freeze
|
||||
|
||||
VALID_SERVICES = [
|
||||
:geocoder_internal,
|
||||
:geocoder_here,
|
||||
:geocoder_google,
|
||||
:geocoder_cache,
|
||||
:geocoder_mapzen,
|
||||
:geocoder_mapbox,
|
||||
:geocoder_tomtom
|
||||
].freeze
|
||||
|
||||
GEOCODER_KEYS = {
|
||||
"heremaps" => :geocoder_here,
|
||||
"google" => :geocoder_google,
|
||||
"mapzen" => :geocoder_mapzen,
|
||||
"mapbox" => :geocoder_mapbox,
|
||||
"tomtom" => :geocoder_tomtom
|
||||
}.freeze
|
||||
|
||||
|
||||
def initialize(username, orgname = nil, redis=$geocoder_metrics)
|
||||
super(username, orgname, redis)
|
||||
end
|
||||
|
||||
protected
|
||||
|
||||
def check_valid_data(service, metric)
|
||||
raise ArgumentError.new('Invalid service') unless VALID_SERVICES.include?(service)
|
||||
raise ArgumentError.new('Invalid metric') unless VALID_METRICS.include?(metric)
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,42 @@
|
||||
require 'active_support/time'
|
||||
require_relative 'service_usage_metrics'
|
||||
|
||||
module CartoDB
|
||||
# The purpose of this class is to encapsulate storage of usage metrics.
|
||||
# This shall be used for billing, quota checking and metrics.
|
||||
class IsolinesUsageMetrics < ServiceUsageMetrics
|
||||
|
||||
VALID_METRICS = [
|
||||
:total_requests,
|
||||
:failed_responses,
|
||||
:success_responses,
|
||||
:empty_responses,
|
||||
:isolines_generated
|
||||
].freeze
|
||||
|
||||
VALID_SERVICES = [
|
||||
:here_isolines,
|
||||
:mapzen_isolines,
|
||||
:mapbox_isolines,
|
||||
:tomtom_isolines
|
||||
].freeze
|
||||
|
||||
ISOLINES_KEYS = {
|
||||
"heremaps" => :here_isolines,
|
||||
"mapzen" => :mapzen_isolines,
|
||||
"mapbox" => :mapbox_isolines,
|
||||
"tomtom" => :tomtom_isolines
|
||||
}.freeze
|
||||
|
||||
def initialize(username, orgname = nil, redis=$geocoder_metrics)
|
||||
super(username, orgname, redis)
|
||||
end
|
||||
|
||||
protected
|
||||
|
||||
def check_valid_data(service, metric)
|
||||
raise ArgumentError.new('Invalid service') unless VALID_SERVICES.include?(service)
|
||||
raise ArgumentError.new('Invalid metric') unless VALID_METRICS.include?(metric)
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,31 @@
|
||||
require 'active_support/time'
|
||||
require_relative 'service_usage_metrics'
|
||||
|
||||
module CartoDB
|
||||
# The purpose of this class is to encapsulate storage of usage metrics.
|
||||
# This shall be used for billing, quota checking and metrics.
|
||||
class ObservatoryGeneralUsageMetrics < ServiceUsageMetrics
|
||||
|
||||
VALID_METRICS = [
|
||||
:total_requests,
|
||||
:failed_responses,
|
||||
:success_responses,
|
||||
:empty_responses
|
||||
].freeze
|
||||
|
||||
VALID_SERVICES = [
|
||||
:obs_general
|
||||
].freeze
|
||||
|
||||
def initialize(username, orgname = nil, redis = $geocoder_metrics)
|
||||
super(username, orgname, redis)
|
||||
end
|
||||
|
||||
protected
|
||||
|
||||
def check_valid_data(service, metric)
|
||||
raise ArgumentError.new('Invalid service') unless VALID_SERVICES.include?(service)
|
||||
raise ArgumentError.new('Invalid metric') unless VALID_METRICS.include?(metric)
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,31 @@
|
||||
require 'active_support/time'
|
||||
require_relative 'service_usage_metrics'
|
||||
|
||||
module CartoDB
|
||||
# The purpose of this class is to encapsulate storage of usage metrics.
|
||||
# This shall be used for billing, quota checking and metrics.
|
||||
class ObservatorySnapshotUsageMetrics < ServiceUsageMetrics
|
||||
|
||||
VALID_METRICS = [
|
||||
:total_requests,
|
||||
:failed_responses,
|
||||
:success_responses,
|
||||
:empty_responses
|
||||
].freeze
|
||||
|
||||
VALID_SERVICES = [
|
||||
:obs_snapshot
|
||||
].freeze
|
||||
|
||||
def initialize(username, orgname = nil, redis = $geocoder_metrics)
|
||||
super(username, orgname, redis)
|
||||
end
|
||||
|
||||
protected
|
||||
|
||||
def check_valid_data(service, metric)
|
||||
raise ArgumentError.new('Invalid service') unless VALID_SERVICES.include?(service)
|
||||
raise ArgumentError.new('Invalid metric') unless VALID_METRICS.include?(metric)
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,39 @@
|
||||
require 'active_support/time'
|
||||
require_relative 'service_usage_metrics'
|
||||
|
||||
module CartoDB
|
||||
# The purpose of this class is to encapsulate storage of usage metrics.
|
||||
# This shall be used for billing, quota checking and metrics.
|
||||
class RoutingUsageMetrics < ServiceUsageMetrics
|
||||
|
||||
VALID_METRICS = [
|
||||
:total_requests,
|
||||
:failed_responses,
|
||||
:success_responses,
|
||||
:empty_responses
|
||||
].freeze
|
||||
|
||||
VALID_SERVICES = [
|
||||
:routing_mapzen,
|
||||
:routing_mapbox,
|
||||
:routing_tomtom
|
||||
].freeze
|
||||
|
||||
ROUTING_KEYS = {
|
||||
"mapzen" => :routing_mapzen,
|
||||
"mapbox" => :routing_mapbox,
|
||||
"tomtom" => :routing_tomtom
|
||||
}.freeze
|
||||
|
||||
def initialize(username, orgname = nil, redis = $geocoder_metrics)
|
||||
super(username, orgname, redis)
|
||||
end
|
||||
|
||||
protected
|
||||
|
||||
def check_valid_data(service, metric)
|
||||
raise ArgumentError.new('Invalid service') unless VALID_SERVICES.include?(service)
|
||||
raise ArgumentError.new('Invalid metric') unless VALID_METRICS.include?(metric)
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,91 @@
|
||||
require 'active_support/time'
|
||||
require_relative '../../../lib/carto/metrics/usage_metrics_interface'
|
||||
|
||||
module CartoDB
|
||||
# The purpose of this class is to encapsulate storage of usage metrics.
|
||||
# This shall be used for billing, quota checking and metrics.
|
||||
class ServiceUsageMetrics < Carto::Metrics::UsageMetricsInterface
|
||||
|
||||
def initialize(username, orgname = nil, redis=$geocoder_metrics)
|
||||
@username = username
|
||||
@orgname = orgname
|
||||
@redis = redis
|
||||
end
|
||||
|
||||
def incr(service, metric, amount = 1, date = DateTime.current)
|
||||
check_valid_data(service, metric)
|
||||
assert_valid_amount(amount)
|
||||
return if amount == 0
|
||||
|
||||
# TODO We could add EXPIRE command to add TTL to the keys
|
||||
if !@orgname.nil?
|
||||
@redis.zincrby("#{org_key_prefix(service, metric, date)}", amount, "#{date_day(date)}")
|
||||
end
|
||||
|
||||
@redis.zincrby("#{user_key_prefix(service, metric, date)}", amount, "#{date_day(date)}")
|
||||
end
|
||||
|
||||
def get(service, metric, date = DateTime.current)
|
||||
check_valid_data(service, metric)
|
||||
|
||||
total = 0
|
||||
if !@orgname.nil?
|
||||
total += @redis.zscore(org_key_prefix(service, metric, date), date_day(date)) || 0
|
||||
else
|
||||
total += @redis.zscore(user_key_prefix(service, metric, date), date_day(date)) || 0
|
||||
end
|
||||
|
||||
total
|
||||
end
|
||||
|
||||
def get_sum_by_date_range(service, metric, date_from, date_to)
|
||||
get_date_range(service, metric, date_from, date_to).values.reduce(:+)
|
||||
end
|
||||
|
||||
def get_date_range(service, metric, date_from, date_to)
|
||||
check_valid_data(service, metric)
|
||||
|
||||
ret = {}
|
||||
month_values = {}
|
||||
(date_from..date_to).each do |date|
|
||||
year_month_key = date_year_month(date)
|
||||
if month_values[year_month_key].nil?
|
||||
key_prefix = @orgname.nil? ? user_key_prefix(service, metric, date) : org_key_prefix(service, metric, date)
|
||||
month_values[year_month_key] = @redis.zrange(key_prefix, 0, -1, with_scores: true).to_h
|
||||
end
|
||||
ret[date] = month_values[year_month_key][date_day(date)] || 0
|
||||
end
|
||||
|
||||
ret
|
||||
end
|
||||
|
||||
protected
|
||||
|
||||
def check_valid_data(_service, _metric)
|
||||
raise NotImplementedError.new("You must implement check_valid_data in your metrics class.")
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def user_key_prefix(service, metric, date)
|
||||
"user:#{@username}:#{service}:#{metric}:#{date_year_month(date)}"
|
||||
end
|
||||
|
||||
def org_key_prefix(service, metric, date)
|
||||
"org:#{@orgname}:#{service}:#{metric}:#{date_year_month(date)}"
|
||||
end
|
||||
|
||||
def date_day(date)
|
||||
date.strftime('%d')
|
||||
end
|
||||
|
||||
def date_year_month(date)
|
||||
date.strftime('%Y%m')
|
||||
end
|
||||
|
||||
def assert_valid_amount(amount)
|
||||
raise ArgumentError.new('Invalid metric amount') if amount.nil? || amount < 0
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,184 @@
|
||||
require_relative '../../lib/service_usage_metrics'
|
||||
require 'mock_redis'
|
||||
require_relative '../../../../spec/rspec_configuration'
|
||||
|
||||
describe CartoDB::ServiceUsageMetrics do
|
||||
|
||||
class DummyServiceUsageMetrics < CartoDB::ServiceUsageMetrics
|
||||
VALID_METRICS = [:dummy_metric].freeze
|
||||
VALID_SERVICES = [:dummy_service].freeze
|
||||
|
||||
def check_valid_data(service, metric)
|
||||
raise ArgumentError.new('Invalid service') unless VALID_SERVICES.include?(service)
|
||||
raise ArgumentError.new('Invalid metric') unless VALID_METRICS.include?(metric)
|
||||
end
|
||||
end
|
||||
|
||||
before(:each) do
|
||||
@redis_mock = MockRedis.new
|
||||
@usage_metrics = DummyServiceUsageMetrics.new('rtorre', 'team', @redis_mock)
|
||||
end
|
||||
|
||||
describe 'Read quota info from redis with zero padding' do
|
||||
|
||||
it 'reads standard zero padded keys' do
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201606', 1543, '01')
|
||||
@usage_metrics.get(:dummy_service, :dummy_metric, Date.new(2016, 6, 1)).should eq 1543
|
||||
end
|
||||
|
||||
it "does not request redis twice when there's no need" do
|
||||
@redis_mock.expects(:zscore).once.with('org:team:dummy_service:dummy_metric:201606', '20').returns(3141592)
|
||||
@usage_metrics.get(:dummy_service, :dummy_metric, Date.new(2016, 6, 20)).should eq 3141592
|
||||
end
|
||||
|
||||
it "returns zero when there's no consumption" do
|
||||
@usage_metrics.get(:dummy_service, :dummy_metric, Date.new(2016, 6, 20)).should eq 0
|
||||
end
|
||||
end
|
||||
|
||||
describe :assert_valid_amount do
|
||||
it 'passes when fed with a positive integer' do
|
||||
@usage_metrics.send(:assert_valid_amount, 42).should eq nil
|
||||
end
|
||||
|
||||
it 'validates that the amount passed cannot be nil' do
|
||||
expect {
|
||||
@usage_metrics.send(:assert_valid_amount, nil)
|
||||
}.to raise_exception(ArgumentError, 'Invalid metric amount')
|
||||
end
|
||||
|
||||
it 'validates that the amount passed cannot be negative' do
|
||||
expect {
|
||||
@usage_metrics.send(:assert_valid_amount, -42)
|
||||
}.to raise_exception(ArgumentError, 'Invalid metric amount')
|
||||
end
|
||||
|
||||
it 'validates that the amount passed can actually be zero' do
|
||||
@usage_metrics.send(:assert_valid_amount, 0).should eq nil
|
||||
end
|
||||
end
|
||||
|
||||
describe :incr do
|
||||
it 'validates that the amount passed can actually be zero' do
|
||||
@usage_metrics.incr(:dummy_service, :dummy_metric, _amount = 0)
|
||||
@usage_metrics.get(:dummy_service, :dummy_metric).should eq 0
|
||||
end
|
||||
end
|
||||
|
||||
describe '#get_sum_by_date_range' do
|
||||
it 'gets a sum of the zscores stored in a given date range' do
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 1, _day = '20')
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 2, _day = '21')
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 3, _day = '22')
|
||||
@usage_metrics.get_sum_by_date_range(:dummy_service,
|
||||
:dummy_metric,
|
||||
Date.new(2017, 3, 20),
|
||||
Date.new(2017, 3, 22)).should eq 6
|
||||
end
|
||||
|
||||
it 'gracefully deals with days without record' do
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 1, _day = '20')
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 2, _day = '21')
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 3, _day = '22')
|
||||
@usage_metrics.get_sum_by_date_range(:dummy_service,
|
||||
:dummy_metric,
|
||||
Date.new(2017, 3, 15),
|
||||
Date.new(2017, 3, 22)).should eq 6
|
||||
end
|
||||
|
||||
it 'gracefully deals with months not stored in redis' do
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 1, _day = '20')
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 2, _day = '21')
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 3, _day = '22')
|
||||
@usage_metrics.get_sum_by_date_range(:dummy_service,
|
||||
:dummy_metric,
|
||||
Date.new(2017, 2, 15),
|
||||
Date.new(2017, 3, 22)).should eq 6
|
||||
end
|
||||
|
||||
it 'performs just one request/month to redis' do
|
||||
@redis_mock.expects(:zrange).twice
|
||||
@usage_metrics.get_sum_by_date_range(:dummy_service,
|
||||
:dummy_metric,
|
||||
Date.new(2017, 2, 15),
|
||||
Date.new(2017, 3, 24))
|
||||
end
|
||||
|
||||
it 'returns zero when there are no records' do
|
||||
@usage_metrics.get_sum_by_date_range(:dummy_service,
|
||||
:dummy_metric,
|
||||
Date.new(2017, 2, 15),
|
||||
Date.new(2017, 3, 22)).should eq 0
|
||||
end
|
||||
end
|
||||
|
||||
describe '#get_date_range' do
|
||||
it 'gets a hash of date => value pairs' do
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 1, _day = '20')
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 2, _day = '21')
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 3, _day = '22')
|
||||
expected = {
|
||||
Date.new(2017, 3, 20) => 1,
|
||||
Date.new(2017, 3, 21) => 2,
|
||||
Date.new(2017, 3, 22) => 3
|
||||
}
|
||||
@usage_metrics.get_date_range(:dummy_service,
|
||||
:dummy_metric,
|
||||
Date.new(2017, 3, 20),
|
||||
Date.new(2017, 3, 22)).should eq expected
|
||||
end
|
||||
|
||||
it 'gracefully deals with days without record' do
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 1, _day = '20')
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 2, _day = '21')
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 3, _day = '22')
|
||||
expected = {
|
||||
Date.new(2017, 3, 18) => 0,
|
||||
Date.new(2017, 3, 19) => 0,
|
||||
Date.new(2017, 3, 20) => 1,
|
||||
Date.new(2017, 3, 21) => 2,
|
||||
Date.new(2017, 3, 22) => 3
|
||||
}
|
||||
@usage_metrics.get_date_range(:dummy_service,
|
||||
:dummy_metric,
|
||||
Date.new(2017, 3, 18),
|
||||
Date.new(2017, 3, 22)).should eq expected
|
||||
end
|
||||
|
||||
it 'gracefully deals with months not stored in redis' do
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 1, _day = '01')
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 2, _day = '02')
|
||||
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 3, _day = '03')
|
||||
expected = {
|
||||
Date.new(2017, 2, 27) => 0,
|
||||
Date.new(2017, 2, 28) => 0,
|
||||
Date.new(2017, 3, 1) => 1,
|
||||
Date.new(2017, 3, 2) => 2,
|
||||
Date.new(2017, 3, 3) => 3
|
||||
}
|
||||
@usage_metrics.get_date_range(:dummy_service,
|
||||
:dummy_metric,
|
||||
Date.new(2017, 2, 27),
|
||||
Date.new(2017, 3, 3)).should eq expected
|
||||
end
|
||||
|
||||
it 'performs just one request/month to redis' do
|
||||
@redis_mock.expects(:zrange).twice
|
||||
@usage_metrics.get_date_range(:dummy_service,
|
||||
:dummy_metric,
|
||||
Date.new(2017, 2, 15),
|
||||
Date.new(2017, 3, 24))
|
||||
end
|
||||
|
||||
it 'returns zero when there are no records' do
|
||||
expected = {
|
||||
Date.new(2017, 2, 28) => 0,
|
||||
Date.new(2017, 3, 1) => 0
|
||||
}
|
||||
@usage_metrics.get_date_range(:dummy_service,
|
||||
:dummy_metric,
|
||||
Date.new(2017, 2, 28),
|
||||
Date.new(2017, 3, 1)).should eq expected
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,16 @@
|
||||
require 'zlib'
|
||||
require_relative './datasources/base'
|
||||
require_relative './datasources/base_oauth'
|
||||
require_relative './datasources/base_file_stream'
|
||||
require_relative './datasources/base_direct_stream'
|
||||
require_relative './datasources/exceptions'
|
||||
require_relative './datasources/url/public_url'
|
||||
require_relative './datasources/url/dropbox'
|
||||
require_relative './datasources/url/gdrive'
|
||||
require_relative './datasources/url/instagram_oauth'
|
||||
require_relative './datasources/url/mailchimp'
|
||||
require_relative './datasources/url/arcgis'
|
||||
require_relative './datasources/search/twitter'
|
||||
require_relative './datasources/datasources_factory'
|
||||
require_relative './datasources/util/csv_file_dumper'
|
||||
require_relative './datasources/decorators/factory'
|
||||
@@ -0,0 +1,141 @@
|
||||
module CartoDB
|
||||
module Datasources
|
||||
class Base
|
||||
|
||||
# .csv
|
||||
FORMAT_CSV = 'csv'
|
||||
# .xls .xlsx
|
||||
FORMAT_EXCEL = 'xls'
|
||||
# .GPX
|
||||
FORMAT_GPX = 'gpx'
|
||||
# .KML
|
||||
FORMAT_KML = 'kml'
|
||||
# .png
|
||||
FORMAT_PNG = 'png'
|
||||
# .jpg .jpeg
|
||||
FORMAT_JPG = 'jpg'
|
||||
# .svg
|
||||
FORMAT_SVG = 'svg'
|
||||
# .zip
|
||||
FORMAT_COMPRESSED = 'zip'
|
||||
|
||||
# If data size cannot be determined, this will be returned as its size in the item metadata
|
||||
NO_CONTENT_SIZE_PROVIDED = 0
|
||||
|
||||
def initialize(*args)
|
||||
@logger = nil
|
||||
end
|
||||
|
||||
# Small helper method to know if metadata includes a valid resource size value or not
|
||||
# @param resource_metadata Hash { :size, ... }
|
||||
# @return bool
|
||||
def has_resource_size?(resource_metadata)
|
||||
resource_metadata[:size] && resource_metadata[:size] > NO_CONTENT_SIZE_PROVIDED
|
||||
end
|
||||
|
||||
# Factory method
|
||||
# @param config {}
|
||||
# @return mixed
|
||||
def get_new(config)
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
# If will provide a url to download the resource, or requires calling get_resource()
|
||||
# @return bool
|
||||
def providers_download_url?
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
# If will provide the url http response code
|
||||
# @return string
|
||||
def get_http_response_code
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
# Perform the listing and return results
|
||||
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
|
||||
# @return [ { :id, :title, :url, :service, :filename, :checksum, :size } ]
|
||||
def get_resources_list(filter={})
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
# Retrieves a resource and returns its contents
|
||||
# @param id string
|
||||
# @return mixed
|
||||
def get_resource(id)
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
# @param id string
|
||||
# @return Hash
|
||||
def get_resource_metadata(id)
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
# Retrieves current filters
|
||||
# @return {}
|
||||
def filter
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
# Sets current filters
|
||||
# @param filter_data {}
|
||||
def filter=(filter_data={})
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
# Log a message
|
||||
# @param message String
|
||||
def log(message)
|
||||
puts message if @logger.nil?
|
||||
@logger.append(message) unless @logger.nil?
|
||||
end
|
||||
|
||||
# @param logger Mixed|nil Set or unset the logger
|
||||
def logger=(logger=nil)
|
||||
@logger = logger
|
||||
end
|
||||
|
||||
# Just return datasource name
|
||||
# @return string
|
||||
def to_s
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
# If this datasource accepts a data import instance
|
||||
# @return Boolean
|
||||
def persists_state_via_data_import?
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
def set_audit_to_completed(table_id = nil)
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
def set_audit_to_failed
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
# @return Hash
|
||||
def get_audit_stats
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
# Stores the data import item instance to use/manipulate it
|
||||
# @param value DataImport
|
||||
def data_import_item=(value)
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
# If true, a single resource id might return >1 subresources (each one spawning a table)
|
||||
# @param id String
|
||||
# @return Bool
|
||||
def multi_resource_import_supported?(id)
|
||||
false
|
||||
end
|
||||
|
||||
private_class_method :new
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,25 @@
|
||||
require_relative 'base'
|
||||
|
||||
module CartoDB
|
||||
module Datasources
|
||||
# Performs streaming from the datasource directly to the caller in batches
|
||||
class BaseDirectStream < Base
|
||||
|
||||
# Initial stream, to be used for container creation (table usually)
|
||||
# @param id string
|
||||
# @return String
|
||||
def initial_stream(id)
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
# @param id string
|
||||
# @return String
|
||||
def stream_resource(id)
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
private_class_method :new
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,17 @@
|
||||
module CartoDB
|
||||
module Datasources
|
||||
# Performs streaming from the datasource to a file
|
||||
class BaseFileStream < Base
|
||||
|
||||
# @param id string
|
||||
# @param stream Stream
|
||||
# @return Integer bytes streamed
|
||||
def stream_resource(id, stream)
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
private_class_method :new
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,93 @@
|
||||
require_relative '../../../importer/lib/importer/unp'
|
||||
|
||||
module CartoDB
|
||||
module Datasources
|
||||
class BaseOAuth < Base
|
||||
|
||||
# At least for CartoDB SaaS, this will be parsed by oauth endpoint to rewrite the url, so must be filled with
|
||||
# e.g. CALLBACK_STATE_DATA_PLACEHOLDER.sub('user', @user.username).sub('service', DATASOURCE_NAME)
|
||||
# And appended anywhere in the querystring used as callback url with param "state=xxxxxx"
|
||||
CALLBACK_STATE_DATA_PLACEHOLDER = '__user__service__'
|
||||
|
||||
attr_reader :config
|
||||
|
||||
def initialize(config, user, mandatory_config_parameters, datasource_name)
|
||||
super
|
||||
mandatory_config_parameters.each { |param| check_config(config, param, datasource_name) }
|
||||
@config = config
|
||||
@user = user
|
||||
end
|
||||
|
||||
# TODO: Helper method to aid with migration of endpoints, can be removed after full AR migration
|
||||
def service_name_for_user(service_name, user)
|
||||
service_name
|
||||
end
|
||||
|
||||
# Return the url to be displayed or sent the user to to authenticate and get authorization code
|
||||
# @param use_callback_flow : bool
|
||||
def get_auth_url(use_callback_flow=true)
|
||||
raise 'To be implemented in child classes'
|
||||
end #get_auth_url
|
||||
|
||||
# Validate authorization code and store token
|
||||
# @param auth_code : string
|
||||
# @return string : Access token
|
||||
def validate_auth_code(auth_code)
|
||||
raise 'To be implemented in child classes'
|
||||
end #validate_auth_code
|
||||
|
||||
# Validates the authorization callback
|
||||
# @param params : mixed
|
||||
def validate_callback(params)
|
||||
raise 'To be implemented in child classes'
|
||||
end #validate_callback
|
||||
|
||||
# Set token
|
||||
# @param token string
|
||||
def token=(token)
|
||||
raise 'To be implemented in child classes'
|
||||
end #token=
|
||||
|
||||
# Retrieve token
|
||||
# @return string | nil
|
||||
def token
|
||||
raise 'To be implemented in child classes'
|
||||
end #token
|
||||
|
||||
# Checks if token is still valid or has been revoked
|
||||
# @return bool
|
||||
def token_valid?
|
||||
raise 'To be implemented in child classes'
|
||||
end #token_valid?
|
||||
|
||||
# Revokes current set token
|
||||
def revoke_token
|
||||
raise 'To be implemented in child classes'
|
||||
end #revoke_token
|
||||
|
||||
private_class_method :new
|
||||
|
||||
protected
|
||||
|
||||
# Calculates a checksum of given input
|
||||
# @param origin string
|
||||
# @return string
|
||||
def checksum_of(origin)
|
||||
#noinspection RubyArgCount
|
||||
Zlib::crc32(origin).to_s
|
||||
end
|
||||
|
||||
def supported_extensions
|
||||
CartoDB::Importer2::Unp::SUPPORTED_FORMATS
|
||||
.concat(CartoDB::Importer2::Unp::COMPRESSED_EXTENSIONS)
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def check_config(config, param, datasource_name)
|
||||
raise MissingConfigurationError.new("missing #{param}", datasource_name) unless config.include?(param)
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,147 @@
|
||||
require_relative './url/arcgis'
|
||||
require_relative './url/dropbox'
|
||||
require_relative './url/box'
|
||||
require_relative './url/gdrive'
|
||||
require_relative './url/instagram_oauth'
|
||||
require_relative './url/mailchimp'
|
||||
require_relative './url/public_url'
|
||||
require_relative 'search/twitter'
|
||||
|
||||
module CartoDB
|
||||
module Datasources
|
||||
class DatasourcesFactory
|
||||
NAME = 'DatasourcesFactory'.freeze
|
||||
|
||||
# in seconds
|
||||
HTTP_CONNECT_TIMEOUT = 60
|
||||
DEFAULT_HTTP_REQUEST_TIMEOUT = 600
|
||||
|
||||
# Retrieve a datasource instance
|
||||
# @param datasource_name string
|
||||
# @param user ::User
|
||||
# @param additional_config Hash
|
||||
# {
|
||||
# :redis_storage => Redis|nil
|
||||
# :ogr2ogr_instance => Ogr2ogr|nil
|
||||
# }
|
||||
# @return mixed
|
||||
# @throws MissingConfigurationError
|
||||
def self.get_datasource(datasource_name, user, additional_config = {})
|
||||
if additional_config[:http_timeout].nil?
|
||||
additional_config[:http_timeout] = DEFAULT_HTTP_REQUEST_TIMEOUT
|
||||
end
|
||||
if additional_config[:http_connect_timeout].nil?
|
||||
additional_config[:http_connect_timeout] = HTTP_CONNECT_TIMEOUT
|
||||
end
|
||||
|
||||
case datasource_name
|
||||
when Url::Dropbox::DATASOURCE_NAME
|
||||
Url::Dropbox.get_new(DatasourcesFactory.config_for(datasource_name, user), user)
|
||||
when Url::Box::DATASOURCE_NAME
|
||||
Url::Box.get_new(DatasourcesFactory.config_for(datasource_name, user), user)
|
||||
when Url::GDrive::DATASOURCE_NAME
|
||||
Url::GDrive.get_new(DatasourcesFactory.config_for(datasource_name, user), user)
|
||||
when Url::InstagramOAuth::DATASOURCE_NAME
|
||||
Url::InstagramOAuth.get_new(DatasourcesFactory.config_for(datasource_name, user), user)
|
||||
when Url::PublicUrl::DATASOURCE_NAME
|
||||
Url::PublicUrl.get_new(additional_config)
|
||||
when Url::ArcGIS::DATASOURCE_NAME
|
||||
Url::ArcGIS.get_new(user)
|
||||
when Url::MailChimp::DATASOURCE_NAME
|
||||
Url::MailChimp.get_new(DatasourcesFactory.config_for(datasource_name, user).merge(additional_config), user)
|
||||
when Search::Twitter::DATASOURCE_NAME
|
||||
Search::Twitter.get_new(DatasourcesFactory.config_for(datasource_name, user), user,
|
||||
additional_config[:redis_storage], additional_config[:user_defined_limits])
|
||||
when nil
|
||||
nil
|
||||
else
|
||||
raise MissingConfigurationError.new("unrecognized datasource #{datasource_name}", NAME)
|
||||
end
|
||||
end
|
||||
|
||||
# Gets all available oauth datasources
|
||||
def self.get_all_oauth_datasources
|
||||
[
|
||||
Url::Dropbox::DATASOURCE_NAME,
|
||||
Url::Box::DATASOURCE_NAME,
|
||||
Url::GDrive::DATASOURCE_NAME,
|
||||
# Url::InstagramOAuth::DATASOURCE_NAME,
|
||||
Url::MailChimp::DATASOURCE_NAME
|
||||
]
|
||||
end
|
||||
|
||||
# Gets the config of a certain datasource
|
||||
# @param datasource_name string
|
||||
# @param user ::User
|
||||
# @return string
|
||||
# @throws MissingConfigurationError
|
||||
def self.config_for(datasource_name, user)
|
||||
config, datasource_supports_custom_config = get_config(datasource_name)
|
||||
|
||||
if datasource_supports_custom_config
|
||||
key = customized_config_key(config, datasource_name, user)
|
||||
|
||||
if key.nil?
|
||||
config[datasource_name][:standard.to_s]
|
||||
else
|
||||
# This code assumes config is ok
|
||||
name_config_map = config[datasource_name]['entity_to_config_map'].select { |u| !u[key].nil? }.first
|
||||
config[datasource_name][:customized.to_s][name_config_map[key]]
|
||||
end
|
||||
else
|
||||
config.fetch(datasource_name)
|
||||
end
|
||||
end
|
||||
|
||||
def self.customized_config?(datasource_name, user)
|
||||
config, datasource_supports_custom_config = get_config(datasource_name)
|
||||
datasource_supports_custom_config && customized_config_key(config, datasource_name, user).present?
|
||||
end
|
||||
|
||||
# Allows to set a custom config (useful for testing)
|
||||
# @param custom_config string
|
||||
def self.set_config(custom_config)
|
||||
@forced_config = custom_config
|
||||
end
|
||||
|
||||
def self.get_config(datasource_name)
|
||||
config_source = @forced_config ? @forced_config : Cartodb.config
|
||||
|
||||
datasource_supports_custom_config = false
|
||||
|
||||
case datasource_name
|
||||
when Url::Dropbox::DATASOURCE_NAME, Url::Box::DATASOURCE_NAME, Url::GDrive::DATASOURCE_NAME, Url::InstagramOAuth::DATASOURCE_NAME,
|
||||
Url::MailChimp::DATASOURCE_NAME
|
||||
config = (config_source[:oauth] rescue nil)
|
||||
config ||= (config_source[:oauth.to_s] rescue nil)
|
||||
when Search::Twitter::DATASOURCE_NAME
|
||||
config = (config_source[:datasource_search] rescue nil)
|
||||
config ||= (config_source[:datasource_search.to_s] rescue nil)
|
||||
datasource_supports_custom_config = true
|
||||
else
|
||||
config = nil
|
||||
end
|
||||
|
||||
if config.nil? || config.empty?
|
||||
raise MissingConfigurationError.new("missing configuration for datasource #{datasource_name}", NAME)
|
||||
end
|
||||
|
||||
[config, datasource_supports_custom_config]
|
||||
end
|
||||
private_class_method :get_config
|
||||
|
||||
def self.customized_config_key(config, datasource_name, user)
|
||||
custom_config_orgs = config[datasource_name].fetch(:customized_orgs_list.to_s, [])
|
||||
custom_config_users = config[datasource_name][:customized_user_list.to_s]
|
||||
|
||||
if user.organization_user? && custom_config_orgs.include?(user.organization.name)
|
||||
user.organization.name
|
||||
elsif custom_config_users.include?(user.username)
|
||||
user.username
|
||||
end
|
||||
end
|
||||
private_class_method :customized_config_key
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
module CartoDB
|
||||
module Datasources
|
||||
module Decorators
|
||||
class BaseDecorator
|
||||
|
||||
# @return bool
|
||||
def decorates_layer?
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
# @param layer Layer|nil
|
||||
# @return bool
|
||||
def layer_eligible?(layer=nil)
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
# @param layer Layer|nil
|
||||
def decorate_layer!(layer=nil)
|
||||
raise 'To be implemented in child classes'
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,27 @@
|
||||
require_relative '../url/instagram_oauth'
|
||||
require_relative './base_decorator'
|
||||
require_relative './instagram_decorator'
|
||||
require_relative './mailchimp_decorator'
|
||||
|
||||
|
||||
module CartoDB
|
||||
module Datasources
|
||||
module Decorators
|
||||
class Factory
|
||||
|
||||
def self.decorator_for(data_import_service_name='')
|
||||
case data_import_service_name
|
||||
when Url::InstagramOAuth::DATASOURCE_NAME
|
||||
Decorators::InstagramDecorator.new
|
||||
when Url::MailChimp::DATASOURCE_NAME
|
||||
Decorators::MailchimpDecorator.new
|
||||
else
|
||||
nil
|
||||
end
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
require_relative './base_decorator'
|
||||
|
||||
module CartoDB
|
||||
module Datasources
|
||||
module Decorators
|
||||
class InstagramDecorator < BaseDecorator
|
||||
|
||||
# @return bool
|
||||
def decorates_layer?
|
||||
true
|
||||
end
|
||||
|
||||
# @param layer Layer|nil
|
||||
# @return bool
|
||||
def layer_eligible?(layer=nil)
|
||||
return false if layer.nil?
|
||||
# Only data/cartodb layers
|
||||
return false unless layer.respond_to?(:data_layer?)
|
||||
layer.data_layer?
|
||||
end
|
||||
|
||||
# @param layer Layer|nil
|
||||
def decorate_layer!(layer=nil)
|
||||
return nil unless layer_eligible?(layer)
|
||||
|
||||
layer.infowindow = {
|
||||
fields: [
|
||||
{position: 0, name: "thumbnail", title: true},
|
||||
{position: 1, name: "caption", title: true},
|
||||
{position: 2, name: "comments_count", title: true},
|
||||
{position: 3, name: "likes_count", title: true},
|
||||
{position: 4, name: "link", title: true}
|
||||
],
|
||||
template_name: "infowindow_header_with_image",
|
||||
alternative_names: {},
|
||||
maxHeight: 275,
|
||||
width: 226,
|
||||
template: ""
|
||||
}
|
||||
nil
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,135 @@
|
||||
require_relative './base_decorator'
|
||||
|
||||
module CartoDB
|
||||
module Datasources
|
||||
module Decorators
|
||||
class MailchimpDecorator < BaseDecorator
|
||||
|
||||
CATEGORY_COLUMN = 'opened'
|
||||
|
||||
CSS_PROPERTIES = {
|
||||
"marker-opacity" => 1,
|
||||
"marker-fill-opacity" => 0.5,
|
||||
"marker-line-color" => "#FFF",
|
||||
"marker-line-width" => 1,
|
||||
"marker-line-opacity" => 1,
|
||||
"marker-placement" => "point",
|
||||
"marker-type" => "ellipse",
|
||||
"marker-width" => 6,
|
||||
"marker-allow-overlap" => true,
|
||||
"marker-comp-op" => 'multiply'
|
||||
}
|
||||
|
||||
CATEGORIES = [
|
||||
{
|
||||
title: true,
|
||||
color: "#A53ED5"
|
||||
},
|
||||
{
|
||||
title: false,
|
||||
color: "#00ceff"
|
||||
}
|
||||
]
|
||||
|
||||
# @return bool
|
||||
def decorates_layer?
|
||||
true
|
||||
end
|
||||
|
||||
# @param layer Layer|nil
|
||||
# @return bool
|
||||
def layer_eligible?(layer=nil)
|
||||
return false if layer.nil?
|
||||
# Only data/cartodb layers
|
||||
return false unless layer.respond_to?(:data_layer?)
|
||||
layer.data_layer?
|
||||
end
|
||||
|
||||
# @param layer Layer|nil
|
||||
def decorate_layer!(layer=nil)
|
||||
return nil unless layer_eligible?(layer)
|
||||
|
||||
enable_category_wizard(layer)
|
||||
enable_category_legend(layer)
|
||||
set_carto_css(layer)
|
||||
|
||||
nil
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def enable_category_wizard(layer)
|
||||
wizard_properties = {
|
||||
type: "category",
|
||||
properties: {
|
||||
property: CATEGORY_COLUMN,
|
||||
"geometry_type" => "point",
|
||||
categories: []
|
||||
}
|
||||
}
|
||||
wizard_properties[:properties].merge!(CSS_PROPERTIES)
|
||||
wizard_properties[:properties][:categories] = CATEGORIES.map do |category|
|
||||
{
|
||||
"title" => category[:title],
|
||||
"title_type" => "boolean",
|
||||
"color" => category[:color],
|
||||
"value_type" => "color"
|
||||
}
|
||||
end
|
||||
|
||||
layer.set_option('wizard_properties', wizard_properties)
|
||||
end
|
||||
|
||||
def enable_category_legend(layer)
|
||||
legend = {
|
||||
"type" => "category",
|
||||
"show_title" => false,
|
||||
"title" => "",
|
||||
"template" => "",
|
||||
"visible" => true
|
||||
}
|
||||
legend[:items] = CATEGORIES.map do |category|
|
||||
{
|
||||
name: category[:title].to_s,
|
||||
visible: true,
|
||||
value: category[:color]
|
||||
}
|
||||
end
|
||||
|
||||
layer.set_option(:legend, legend)
|
||||
end
|
||||
|
||||
def set_carto_css(layer)
|
||||
matches = layer.options['tile_style'].match(/^#(.*) \{/)
|
||||
unless matches.nil?
|
||||
css_selector = "##{matches[1]}"
|
||||
css_properties = CSS_PROPERTIES.map{|property, value| " #{property}: #{value};"}.join("\n")
|
||||
|
||||
carto_css = []
|
||||
carto_css << "#{css_selector} {"
|
||||
carto_css << css_properties
|
||||
carto_css << " [zoom>4] {"
|
||||
carto_css << " marker-width: 7;"
|
||||
carto_css << " }"
|
||||
carto_css << " [zoom>5] {"
|
||||
carto_css << " marker-width: 8;"
|
||||
carto_css << " }"
|
||||
carto_css << " [zoom>6] {"
|
||||
carto_css << " marker-width: 9;"
|
||||
carto_css << " }"
|
||||
carto_css << "}"
|
||||
|
||||
CATEGORIES.each do |category|
|
||||
carto_css << "#{css_selector}[#{CATEGORY_COLUMN}=#{category[:title]}] {"
|
||||
carto_css << " marker-fill: #{category[:color]};"
|
||||
carto_css << "}"
|
||||
end
|
||||
|
||||
layer.set_option('tile_style', carto_css.join("\n"))
|
||||
layer.set_option('tile_style_custom', false)
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,66 @@
|
||||
module CartoDB
|
||||
module Datasources
|
||||
|
||||
# Remember to add new errors to:
|
||||
# config/initializers/carto_db.rb
|
||||
# services/importer/lib/importer/exceptions.rb
|
||||
|
||||
class DatasourceBaseError < StandardError
|
||||
|
||||
UNKNOWN_SERVICE = 'UNKNOWN'.freeze
|
||||
|
||||
attr_reader :service_name
|
||||
|
||||
def initialize(message = 'General error', service = UNKNOWN_SERVICE, username = nil)
|
||||
@service_name = service
|
||||
message = "#{message}"
|
||||
message << " @ #{@service_name}" if @service_name != UNKNOWN_SERVICE
|
||||
message << " User: #{username}" unless username.nil?
|
||||
super(message)
|
||||
end
|
||||
end
|
||||
|
||||
class AuthError < DatasourceBaseError; end
|
||||
# This exception is ONLY throwed if oauth token is wrong or expired, and should be deleted if exists
|
||||
class TokenExpiredOrInvalidError < AuthError; end
|
||||
class InvalidServiceError < DatasourceBaseError; end
|
||||
class DataDownloadError < DatasourceBaseError; end
|
||||
class UnsupportedOperationError < DatasourceBaseError; end
|
||||
class NotFoundDownloadError < DatasourceBaseError; end
|
||||
class MissingConfigurationError < DatasourceBaseError; end
|
||||
class UninitializedError < DatasourceBaseError; end
|
||||
class NoResultsError < DatasourceBaseError; end
|
||||
class ParameterError < DatasourceBaseError; end
|
||||
|
||||
class OutOfQuotaError < DatasourceBaseError; end
|
||||
class InvalidInputDataError < DatasourceBaseError; end
|
||||
class ResponseError < DatasourceBaseError; end
|
||||
class ExternalServiceError < DatasourceBaseError; end
|
||||
|
||||
class GNIPServiceError < ExternalServiceError; end
|
||||
|
||||
class ServiceDisabledError < DatasourceBaseError
|
||||
def initialize(service = UNKNOWN_SERVICE, username = nil)
|
||||
super("Service disabled", service, username)
|
||||
end
|
||||
end
|
||||
|
||||
class DataDownloadTimeoutError < DatasourceBaseError
|
||||
def initialize(service = UNKNOWN_SERVICE, username = nil)
|
||||
super("Data download timed out. Check the source is not running slow and/or try again.", service, username)
|
||||
end
|
||||
end
|
||||
|
||||
class ExternalServiceTimeoutError < DatasourceBaseError
|
||||
def initialize(service = UNKNOWN_SERVICE, username = nil)
|
||||
super("External service timed out. Check the source is not running slow and/or try again.", service, username)
|
||||
end
|
||||
end
|
||||
|
||||
class DatasourcePermissionError < DatasourceBaseError; end
|
||||
class DropboxPermissionError < DatasourcePermissionError; end
|
||||
class BoxPermissionError < DatasourcePermissionError; end
|
||||
|
||||
class GDriveNoExternalAppsAllowedError < DatasourceBaseError; end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,597 @@
|
||||
require 'json'
|
||||
|
||||
require_relative '../util/csv_file_dumper'
|
||||
|
||||
require_relative '../../../../twitter-search/twitter-search'
|
||||
require_relative '../../../../../lib/cartodb/logger'
|
||||
require_relative '../base_file_stream'
|
||||
|
||||
module CartoDB
|
||||
module Datasources
|
||||
module Search
|
||||
|
||||
# NOTE: 'redis_storage' is only sent in normal imports, not at OAuth or Synchronizations,
|
||||
# as this datasource is not intended to be used in such.
|
||||
class Twitter < BaseFileStream
|
||||
|
||||
# Required for all datasources
|
||||
DATASOURCE_NAME = 'twitter_search'
|
||||
|
||||
NO_TOTAL_RESULTS = -1
|
||||
|
||||
MAX_CATEGORIES = 4
|
||||
|
||||
DEBUG_FLAG = false
|
||||
|
||||
# Used for each query page size, not as total
|
||||
FILTER_MAXRESULTS = :maxResults
|
||||
FILTER_FROMDATE = :fromDate
|
||||
FILTER_TODATE = :toDate
|
||||
FILTER_CATEGORIES = :categories
|
||||
FILTER_TOTAL_RESULTS = :totalResults
|
||||
|
||||
USER_LIMITS_FILTER_CREDITS = :twitter_credits_limit
|
||||
|
||||
CATEGORY_NAME_KEY = :name
|
||||
CATEGORY_TERMS_KEY = :terms
|
||||
|
||||
GEO_SEARCH_FILTER = 'has:geo'
|
||||
PROFILE_GEO_SEARCH_FILTER = 'has:profile_geo'
|
||||
OR_SEARCH_FILTER = 'OR'
|
||||
|
||||
# Seconds to substract from current time as threshold to consider a time
|
||||
# as "now or from the future" upon date filter build
|
||||
TIMEZONE_THRESHOLD = 60
|
||||
|
||||
# Gnip's 30 limit minus 'has:geo' one
|
||||
MAX_SEARCH_TERMS = 30 - 1
|
||||
|
||||
MAX_QUERY_SIZE = 2048
|
||||
|
||||
MAX_TABLE_NAME_SIZE = 30
|
||||
|
||||
# Constructor
|
||||
# @param config Array
|
||||
# [
|
||||
# 'auth_required'
|
||||
# 'username'
|
||||
# 'password'
|
||||
# 'search_url'
|
||||
# ]
|
||||
# @param user ::User
|
||||
# @param redis_storage Redis|nil (optional)
|
||||
# @param user_defined_limits Hash|nil (optional)
|
||||
# @throws UninitializedError
|
||||
def initialize(config, user, redis_storage = nil, user_defined_limits={})
|
||||
@service_name = DATASOURCE_NAME
|
||||
@filters = Hash.new
|
||||
|
||||
raise UninitializedError.new('missing user instance', DATASOURCE_NAME) if user.nil?
|
||||
raise MissingConfigurationError.new('missing auth_required', DATASOURCE_NAME) unless config.include?('auth_required')
|
||||
raise MissingConfigurationError.new('missing username', DATASOURCE_NAME) unless config.include?('username')
|
||||
raise MissingConfigurationError.new('missing password', DATASOURCE_NAME) unless config.include?('password')
|
||||
raise MissingConfigurationError.new('missing search_url for GNIP API', DATASOURCE_NAME) unless config.include?('search_url')
|
||||
|
||||
@user_defined_limits = user_defined_limits
|
||||
|
||||
@search_api_config = {
|
||||
TwitterSearch::SearchAPI::CONFIG_AUTH_REQUIRED => config['auth_required'],
|
||||
TwitterSearch::SearchAPI::CONFIG_AUTH_USERNAME => config['username'],
|
||||
TwitterSearch::SearchAPI::CONFIG_AUTH_PASSWORD => config['password'],
|
||||
TwitterSearch::SearchAPI::CONFIG_SEARCH_URL => config['search_url'],
|
||||
TwitterSearch::SearchAPI::CONFIG_REDIS_RL_ACTIVE => config.fetch('ratelimit_active', nil),
|
||||
TwitterSearch::SearchAPI::CONFIG_REDIS_RL_MAX_CONCURRENCY => config.fetch('ratelimit_concurrency', nil),
|
||||
TwitterSearch::SearchAPI::CONFIG_REDIS_RL_TTL => config.fetch('ratelimit_ttl', nil),
|
||||
TwitterSearch::SearchAPI::CONFIG_REDIS_RL_WAIT_SECS => config.fetch('ratelimit_wait_secs', nil)
|
||||
}
|
||||
@redis_storage = redis_storage
|
||||
|
||||
@csv_dumper = CSVFileDumper.new(TwitterSearch::JSONToCSVConverter.new, DEBUG_FLAG)
|
||||
|
||||
@user = user
|
||||
@data_import_item = nil
|
||||
|
||||
@logger = nil
|
||||
@used_quota = 0
|
||||
@user_semaphore = Mutex.new
|
||||
end
|
||||
|
||||
# Factory method
|
||||
# @param config {}
|
||||
# @param user ::User
|
||||
# @param redis_storage Redis|nil
|
||||
# @param user_defined_limits Hash|nil
|
||||
# @return CartoDB::Datasources::Search::TwitterSearch
|
||||
def self.get_new(config, user, redis_storage = nil, user_defined_limits={})
|
||||
return new(config, user, redis_storage, user_defined_limits)
|
||||
end
|
||||
|
||||
# If will provide a url to download the resource, or requires calling get_resource()
|
||||
# @return bool
|
||||
def providers_download_url?
|
||||
false
|
||||
end
|
||||
|
||||
# Perform the listing and return results
|
||||
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
|
||||
# @return [ { :id, :title, :url, :service } ]
|
||||
def get_resources_list(filter=[])
|
||||
filter
|
||||
end
|
||||
|
||||
# @param id string
|
||||
# @param stream Stream
|
||||
# @return Integer bytes streamed
|
||||
def stream_resource(id, stream)
|
||||
unless has_enough_quota?(@user)
|
||||
raise OutOfQuotaError.new("#{@user.username} out of quota for tweets", DATASOURCE_NAME)
|
||||
end
|
||||
raise ServiceDisabledError.new(DATASOURCE_NAME, @user.username) unless is_service_enabled?(@user)
|
||||
|
||||
fields_from(id)
|
||||
|
||||
do_search(@search_api_config, @redis_storage, @filters, stream)
|
||||
end
|
||||
|
||||
# Retrieves a resource and returns its contents
|
||||
# @param id string Will contain a stringified JSON
|
||||
# @return mixed
|
||||
# @throws ServiceDisabledError
|
||||
# @throws OutOfQuotaError
|
||||
# @throws ParameterError
|
||||
# @deprecated Use stream_resource instead
|
||||
def get_resource(id)
|
||||
unless has_enough_quota?(@user)
|
||||
raise OutOfQuotaError.new("#{@user.username} out of quota for tweets", DATASOURCE_NAME)
|
||||
end
|
||||
raise ServiceDisabledError.new(DATASOURCE_NAME, @user.username) unless is_service_enabled?(@user)
|
||||
|
||||
fields_from(id)
|
||||
|
||||
do_search(@search_api_config, @redis_storage, @filters, stream = nil)
|
||||
end
|
||||
|
||||
# @param id string
|
||||
# @return Hash
|
||||
def get_resource_metadata(id)
|
||||
fields_from(id)
|
||||
{
|
||||
id: id,
|
||||
title: DATASOURCE_NAME,
|
||||
url: nil,
|
||||
service: DATASOURCE_NAME,
|
||||
checksum: nil,
|
||||
size: 0,
|
||||
filename: "#{table_name}.csv"
|
||||
}
|
||||
end
|
||||
|
||||
# Retrieves current filters. Unused as here there's no get_resources_list
|
||||
# @return {}
|
||||
def filter
|
||||
{}
|
||||
end
|
||||
|
||||
# Sets current filters. Unused as here there's no get_resources_list
|
||||
# @param filter_data {}
|
||||
def filter=(filter_data=[])
|
||||
filter_data
|
||||
end
|
||||
|
||||
# Hide sensitive fields
|
||||
def to_s
|
||||
"<CartoDB::Datasources::Search::Twitter @user=#{@user.username} @filters=#{@filters} @search_api_config=#{search_api_config_public_values}>"
|
||||
end
|
||||
|
||||
# If this datasource accepts a data import instance
|
||||
# @return Boolean
|
||||
def persists_state_via_data_import?
|
||||
true
|
||||
end
|
||||
|
||||
# Stores the data import item instance to use/manipulate it
|
||||
# @param value DataImport
|
||||
def data_import_item=(value)
|
||||
@data_import_item = value
|
||||
end
|
||||
|
||||
def set_audit_to_completed(table_id = nil)
|
||||
entry = audit_entry.class.where(data_import_id:@data_import_item.id).first
|
||||
raise DatasourceBaseError.new("Couldn't fetch SearchTweet entry for data import #{@data_import_item.id}", \
|
||||
DATASOURCE_NAME) if entry.nil?
|
||||
|
||||
entry.set_complete_state
|
||||
entry.table_id = table_id unless table_id.nil?
|
||||
entry.save
|
||||
end
|
||||
|
||||
def set_audit_to_failed
|
||||
entry = audit_entry.class.where(data_import_id:@data_import_item.id).first
|
||||
raise DatasourceBaseError.new("Couldn't fetch SearchTweet entry for data import #{@data_import_item.id}", \
|
||||
DATASOURCE_NAME) if entry.nil?
|
||||
|
||||
entry.set_failed_state
|
||||
entry.save
|
||||
end
|
||||
|
||||
# @return Hash
|
||||
def get_audit_stats
|
||||
entry = audit_entry.class.where(data_import_id:@data_import_item.id).first
|
||||
raise DatasourceBaseError.new("Couldn't fetch SearchTweet entry for data import #{@data_import_item.id}", \
|
||||
DATASOURCE_NAME) if entry.nil?
|
||||
{ :retrieved_items => entry.retrieved_items }
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
# Used at specs
|
||||
attr_accessor :search_api_config, :csv_dumper
|
||||
attr_reader :data_import_item
|
||||
|
||||
def search_api_config_public_values
|
||||
{
|
||||
TwitterSearch::SearchAPI::CONFIG_AUTH_REQUIRED =>
|
||||
@search_api_config[TwitterSearch::SearchAPI::CONFIG_AUTH_REQUIRED],
|
||||
TwitterSearch::SearchAPI::CONFIG_AUTH_USERNAME =>
|
||||
@search_api_config[TwitterSearch::SearchAPI::CONFIG_AUTH_USERNAME],
|
||||
TwitterSearch::SearchAPI::CONFIG_SEARCH_URL =>
|
||||
@search_api_config[TwitterSearch::SearchAPI::CONFIG_SEARCH_URL]
|
||||
}
|
||||
end
|
||||
|
||||
# Returns if the user set a maximum credits to use
|
||||
# @return Integer
|
||||
def twitter_credit_limits
|
||||
@user_defined_limits.fetch(USER_LIMITS_FILTER_CREDITS, 0)
|
||||
end
|
||||
|
||||
# Wraps check of specified user limit or not (to use instead his max quota)
|
||||
# @return Integer
|
||||
def remaining_quota
|
||||
twitter_credit_limits > 0 ? [@user.remaining_twitter_quota, twitter_credit_limits].min
|
||||
: @user.remaining_twitter_quota
|
||||
end
|
||||
|
||||
def table_name
|
||||
terms_fragment = @filters[FILTER_CATEGORIES].map { |category|
|
||||
clean_category(category[CATEGORY_TERMS_KEY]).gsub(/[^0-9a-z,]/i, '').gsub(/[,]/i, '_')
|
||||
}.join('_').slice(0,MAX_TABLE_NAME_SIZE)
|
||||
|
||||
"twitter_#{terms_fragment}"
|
||||
end
|
||||
|
||||
def clean_category(category)
|
||||
category.gsub(" (#{GEO_SEARCH_FILTER} OR #{PROFILE_GEO_SEARCH_FILTER})", '')
|
||||
.gsub(" #{OR_SEARCH_FILTER} ", ', ')
|
||||
.gsub(/^\(/, '')
|
||||
.gsub(/\)$/, '')
|
||||
end
|
||||
|
||||
def fields_from(id)
|
||||
return unless @filters.count == 0
|
||||
|
||||
fields = ::JSON.parse(id, symbolize_names: true)
|
||||
|
||||
@filters[FILTER_CATEGORIES] = build_queries_from_fields(fields)
|
||||
|
||||
if @filters[FILTER_CATEGORIES].size > MAX_CATEGORIES
|
||||
raise ParameterError.new("Max allowed categories are #{FILTER_CATEGORIES}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
@filters[FILTER_FROMDATE] = build_date_from_fields(fields, 'from')
|
||||
@filters[FILTER_TODATE] = build_date_from_fields(fields, 'to')
|
||||
@filters[FILTER_MAXRESULTS] = build_maxresults_field(@user)
|
||||
@filters[FILTER_TOTAL_RESULTS] = build_total_results_field(@user)
|
||||
end
|
||||
|
||||
# Signature must be like: .report_message('Import error', 'error', error_info: stacktrace)
|
||||
def report_error(message, additional_data)
|
||||
log("Error: #{message} Additional Info: #{additional_data}")
|
||||
CartoDB::Logger.error(message: message, error_info: additional_data)
|
||||
end
|
||||
|
||||
# @param api_config Hash
|
||||
# @param redis_storage Mixed
|
||||
# @param filters Hash
|
||||
# @param stream IO
|
||||
# @return Mixed The data
|
||||
def do_search(api_config, redis_storage, filters, stream)
|
||||
threads = {}
|
||||
base_filters = filters.select { |k, v| k != FILTER_CATEGORIES }
|
||||
|
||||
category_totals = {}
|
||||
dumper_additional_fields = {}
|
||||
filters[FILTER_CATEGORIES].each { |category|
|
||||
dumper_additional_fields[category[CATEGORY_NAME_KEY]] = {
|
||||
category_name: category[CATEGORY_NAME_KEY],
|
||||
category_terms: clean_category(category[CATEGORY_TERMS_KEY])
|
||||
}
|
||||
@csv_dumper.begin_dump(category[CATEGORY_NAME_KEY])
|
||||
}
|
||||
@csv_dumper.additional_fields = dumper_additional_fields
|
||||
|
||||
log("Searching #{filters[FILTER_CATEGORIES].length} categories")
|
||||
|
||||
filters[FILTER_CATEGORIES].each { |category|
|
||||
# If all threads are created at the same time, redis semaphore inside search_api
|
||||
# might not yet have new value, so introduce a small delay on each thread creation
|
||||
sleep(0.1)
|
||||
threads[category[CATEGORY_NAME_KEY]] = Thread.new {
|
||||
api = TwitterSearch::SearchAPI.new(api_config, redis_storage, @csv_dumper)
|
||||
# Dumps happen inside upon each block response
|
||||
total_results = search_by_category(api, base_filters, category)
|
||||
category_totals[category[CATEGORY_NAME_KEY]] = total_results
|
||||
}
|
||||
}
|
||||
threads.each {|key, thread|
|
||||
thread.join
|
||||
}
|
||||
|
||||
# INFO: For now we don't treat as error a no results scenario, else use:
|
||||
# raise NoResultsError.new if category_totals.values.inject(:+) == 0
|
||||
|
||||
filters[FILTER_CATEGORIES].each { |category|
|
||||
@csv_dumper.end_dump(category[CATEGORY_NAME_KEY])
|
||||
}
|
||||
streamed_size = @csv_dumper.merge_dumps_into_stream(dumper_additional_fields.keys, stream)
|
||||
|
||||
log("Temp files:\n#{@csv_dumper.file_paths}")
|
||||
log("#{@csv_dumper.original_file_paths}\n#{@csv_dumper.headers_path}")
|
||||
|
||||
if twitter_credit_limits > 0 || !@user.soft_twitter_datasource_limit
|
||||
if (remaining_quota - @used_quota) < 0
|
||||
# Make sure we don't charge extra tweets (even if we "lose" charging a block or two of tweets)
|
||||
@used_quota = remaining_quota
|
||||
end
|
||||
end
|
||||
|
||||
# remaining quota is calc. on the fly based on audits/imports
|
||||
save_audit(@user, @data_import_item, @used_quota)
|
||||
|
||||
streamed_size
|
||||
end
|
||||
|
||||
def search_by_category(api, base_filters, category)
|
||||
api.params = base_filters
|
||||
|
||||
exception = nil
|
||||
next_results_cursor = nil
|
||||
total_results = 0
|
||||
|
||||
begin
|
||||
exception = nil
|
||||
out_of_quota = false
|
||||
|
||||
@user_semaphore.synchronize {
|
||||
# Credit limits must be honoured above soft limit
|
||||
if twitter_credit_limits > 0 || !@user.soft_twitter_datasource_limit
|
||||
if remaining_quota - @used_quota <= 0
|
||||
out_of_quota = true
|
||||
next_results_cursor = nil
|
||||
end
|
||||
end
|
||||
}
|
||||
|
||||
unless out_of_quota
|
||||
api.query_param = category[CATEGORY_TERMS_KEY]
|
||||
begin
|
||||
results_page = api.fetch_results(next_results_cursor)
|
||||
rescue TwitterSearch::TwitterHTTPException => e
|
||||
exception = e
|
||||
report_error(e.to_s, e.backtrace)
|
||||
# Stop gracefully to not break whole import process
|
||||
results_page = {
|
||||
results: [],
|
||||
next: nil
|
||||
}
|
||||
end
|
||||
|
||||
dumped_items_count = @csv_dumper.dump(category[CATEGORY_NAME_KEY], results_page[:results])
|
||||
next_results_cursor = results_page[:next].nil? ? nil : results_page[:next]
|
||||
|
||||
@user_semaphore.synchronize {
|
||||
@used_quota += dumped_items_count
|
||||
}
|
||||
|
||||
total_results += dumped_items_count
|
||||
end
|
||||
end while (!next_results_cursor.nil? && !out_of_quota && !exception)
|
||||
|
||||
log("'#{category[CATEGORY_NAME_KEY]}' got #{total_results} results")
|
||||
log("Got exception at '#{category[CATEGORY_NAME_KEY]}': #{exception.inspect}") if exception
|
||||
|
||||
# If fails on the first request, bubble up the error, else will return as many tweets as possible
|
||||
if !exception.nil? && total_results == 0
|
||||
log("ERROR: 0 results & exception: #{exception} (HTTP #{exception.http_code}) #{exception.additional_data}")
|
||||
# @see http://support.gnip.com/apis/search_api/api_reference.html
|
||||
if exception.http_code == 422 && exception.additional_data =~ /request usage cap exceeded/i
|
||||
raise OutOfQuotaError.new(exception.to_s, DATASOURCE_NAME)
|
||||
end
|
||||
if [401, 404].include?(exception.http_code)
|
||||
raise MissingConfigurationError.new(exception.to_s, DATASOURCE_NAME)
|
||||
end
|
||||
if [400, 422].include?(exception.http_code)
|
||||
raise InvalidInputDataError.new(exception.to_s, DATASOURCE_NAME)
|
||||
end
|
||||
if exception.http_code == 429
|
||||
raise ResponseError.new(exception.to_s, DATASOURCE_NAME)
|
||||
end
|
||||
if exception.http_code >= 500 && exception.http_code < 600
|
||||
raise GNIPServiceError.new(exception.to_s, DATASOURCE_NAME)
|
||||
end
|
||||
raise DatasourceBaseError.new(exception.to_s, DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
total_results
|
||||
end
|
||||
|
||||
def build_date_from_fields(fields, date_type)
|
||||
raise ParameterError.new('missing dates', DATASOURCE_NAME) \
|
||||
if fields[:dates].nil?
|
||||
|
||||
case date_type
|
||||
when 'from'
|
||||
date_sym = :fromDate
|
||||
hour_sym = :fromHour
|
||||
min_sym = :fromMin
|
||||
when 'to'
|
||||
date_sym = :toDate
|
||||
hour_sym = :toHour
|
||||
min_sym = :toMin
|
||||
else
|
||||
raise ParameterError.new("unknown date type #{date_type}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
if fields[:dates][date_sym].nil? || fields[:dates][hour_sym].nil? || fields[:dates][min_sym].nil?
|
||||
date = nil
|
||||
else
|
||||
# Sent by JS in minutes
|
||||
timezone = fields[:dates][:user_timezone].nil? ? 0 : fields[:dates][:user_timezone].to_i
|
||||
begin
|
||||
year, month, day = fields[:dates][date_sym].split('-')
|
||||
timezoned_date = Time.gm(year, month, day, fields[:dates][hour_sym], fields[:dates][min_sym])
|
||||
rescue ArgumentError
|
||||
raise ParameterError.new('Invalid date format', DATASOURCE_NAME)
|
||||
end
|
||||
timezoned_date += timezone*60
|
||||
|
||||
# Gnip doesn't allows searches "in the future"
|
||||
date = timezoned_date >= (Time.now - TIMEZONE_THRESHOLD).utc ? nil : timezoned_date.strftime("%Y%m%d%H%M")
|
||||
end
|
||||
|
||||
date
|
||||
end
|
||||
|
||||
def build_queries_from_fields(fields)
|
||||
raise ParameterError.new('missing categories', DATASOURCE_NAME) \
|
||||
if fields[:categories].nil? || fields[:categories].empty?
|
||||
|
||||
queries = []
|
||||
fields[:categories].each { |category|
|
||||
raise ParameterError.new('missing category', DATASOURCE_NAME) if category[:category].nil?
|
||||
raise ParameterError.new('missing terms', DATASOURCE_NAME) if category[:terms].nil?
|
||||
|
||||
# Gnip limitation
|
||||
if category[:terms].count > MAX_SEARCH_TERMS
|
||||
category[:terms] = category[:terms].slice(0, MAX_SEARCH_TERMS)
|
||||
end
|
||||
|
||||
category[:terms] = sanitize_terms(category[:terms])
|
||||
|
||||
query = {
|
||||
CATEGORY_NAME_KEY => category[:category].to_s,
|
||||
CATEGORY_TERMS_KEY => ''
|
||||
}
|
||||
|
||||
unless category[:terms].count == 0
|
||||
query[CATEGORY_TERMS_KEY] << '('
|
||||
query[CATEGORY_TERMS_KEY] << category[:terms].join(' OR ')
|
||||
query[CATEGORY_TERMS_KEY] << ") (#{GEO_SEARCH_FILTER} OR #{PROFILE_GEO_SEARCH_FILTER})"
|
||||
end
|
||||
|
||||
if query[CATEGORY_TERMS_KEY].length > MAX_QUERY_SIZE
|
||||
raise ParameterError.new("Obtained search query is bigger than #{MAX_QUERY_SIZE} chars", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
queries << query
|
||||
}
|
||||
queries
|
||||
end
|
||||
|
||||
# @param terms_list Array
|
||||
def sanitize_terms(terms_list)
|
||||
terms_list.map{ |term|
|
||||
# Remove unwanted stuff
|
||||
sanitized = term.to_s.gsub(/^ /, '').gsub(/ $/, '').gsub('"', '')
|
||||
# Quote if needed
|
||||
if sanitized.gsub(/[a-z0-9@#]/i,'') != ''
|
||||
sanitized = '"' + sanitized + '"'
|
||||
end
|
||||
sanitized.length == 0 ? nil : sanitized
|
||||
}.compact
|
||||
end
|
||||
|
||||
# Max results per page
|
||||
# @param user ::User
|
||||
def build_maxresults_field(user)
|
||||
if twitter_credit_limits > 0
|
||||
[remaining_quota, TwitterSearch::SearchAPI::MAX_PAGE_RESULTS].min
|
||||
else
|
||||
# user about to hit quota?
|
||||
if remaining_quota < TwitterSearch::SearchAPI::MAX_PAGE_RESULTS
|
||||
if user.soft_twitter_datasource_limit
|
||||
# But can go beyond limits
|
||||
TwitterSearch::SearchAPI::MAX_PAGE_RESULTS
|
||||
else
|
||||
remaining_quota
|
||||
end
|
||||
else
|
||||
TwitterSearch::SearchAPI::MAX_PAGE_RESULTS
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
# Max total results
|
||||
# @param user ::User
|
||||
def build_total_results_field(user)
|
||||
if twitter_credit_limits == 0 && user.soft_twitter_datasource_limit
|
||||
NO_TOTAL_RESULTS
|
||||
else
|
||||
remaining_quota
|
||||
end
|
||||
end
|
||||
|
||||
# @param user ::User
|
||||
def is_service_enabled?(user)
|
||||
if !user.organization.nil?
|
||||
enabled = user.organization.twitter_datasource_enabled
|
||||
if enabled
|
||||
user.twitter_datasource_enabled
|
||||
else
|
||||
# If disabled org-wide, disabled for everyone
|
||||
false
|
||||
end
|
||||
else
|
||||
user.twitter_datasource_enabled
|
||||
end
|
||||
end
|
||||
|
||||
# @param user ::User
|
||||
# @return boolean
|
||||
def has_enough_quota?(user)
|
||||
# As this is used to disallow searches (and throw exceptions) don't use here user limits
|
||||
user.soft_twitter_datasource_limit || (user.remaining_twitter_quota > 0)
|
||||
end
|
||||
|
||||
# @param user ::User
|
||||
# @param data_import_item DataImport
|
||||
# @param retrieved_items_count Integer
|
||||
def save_audit(user, data_import_item, retrieved_items_count)
|
||||
entry = audit_entry
|
||||
entry.set_importing_state
|
||||
entry.user_id = user.id
|
||||
entry.data_import_id = data_import_item.id
|
||||
entry.service_item_id = data_import_item.service_item_id
|
||||
entry.retrieved_items = retrieved_items_count
|
||||
entry.save
|
||||
end
|
||||
|
||||
# Call this inside specs to override returned class
|
||||
# @param override_class SearchTweet|nil (optional)
|
||||
# @return SearchTweet
|
||||
def audit_entry(override_class = nil)
|
||||
if @audit_entry.nil?
|
||||
if override_class.nil?
|
||||
require_relative '../../../../../app/models/search_tweet'
|
||||
@audit_entry = ::SearchTweet.new
|
||||
else
|
||||
@audit_entry = override_class.new
|
||||
end
|
||||
end
|
||||
@audit_entry
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,540 @@
|
||||
require 'json'
|
||||
require 'addressable/uri'
|
||||
require_relative '../base_direct_stream'
|
||||
require_relative '../../../../../lib/carto/http/client'
|
||||
|
||||
module CartoDB
|
||||
module Datasources
|
||||
module Url
|
||||
|
||||
class ArcGIS < BaseDirectStream
|
||||
|
||||
# Required for all datasources
|
||||
DATASOURCE_NAME = 'arcgis'
|
||||
|
||||
ARCGIS_API_LIKE_URL_RE = /\/rest\/services/i
|
||||
|
||||
METADATA_URL = '%s?f=json'
|
||||
FEATURE_IDS_URL = '%s/query?where=1%%3D1&returnIdsOnly=true&f=json'
|
||||
FEATURE_DATA_POST_URL = '%s/query'
|
||||
LAYERS_URL = '%s/layers?f=json'
|
||||
|
||||
MINIMUM_SUPPORTED_VERSION = 10.1
|
||||
|
||||
# In seconds, for connecting
|
||||
HTTP_CONNECTION_TIMEOUT = 60
|
||||
# In seconds, for the full request
|
||||
HTTP_TIMEOUT = 60
|
||||
|
||||
# Amount to multiply or divide
|
||||
BLOCK_FACTOR = 2
|
||||
MIN_BLOCK_SIZE = 1
|
||||
# GeoJSON can get too big in memory, or ArcGIS have mem problems, so keep reasonable number
|
||||
MAX_BLOCK_SIZE = 100
|
||||
# In seconds, use 0 to disable
|
||||
BLOCK_SLEEP_TIME = 1
|
||||
|
||||
# Each retry will be after SLEEP_REQUEST_TIME^(current_retries_count). Set to 0 to disable retrying
|
||||
MAX_RETRIES = 0
|
||||
SLEEP_REQUEST_TIME = 3
|
||||
SKIP_FAILED_IDS = true
|
||||
|
||||
# Used to display more data only (for local debugging purposes)
|
||||
DEBUG = false
|
||||
|
||||
VECTOR_LAYER_TYPE = 'Feature Layer'.freeze
|
||||
OID_FIELD_TYPE = 'esriFieldTypeOID'.freeze
|
||||
|
||||
attr_reader :metadata
|
||||
|
||||
# Constructor
|
||||
# @param user ::User
|
||||
def initialize(user)
|
||||
super
|
||||
@service_name = DATASOURCE_NAME
|
||||
|
||||
# Fields:
|
||||
# @metadata = {
|
||||
# arcgis_version: nil,
|
||||
# name: nil,
|
||||
# description: nil,
|
||||
# type: nil,
|
||||
# geometry_type: nil,
|
||||
# copyright: nil,
|
||||
# fields: [],
|
||||
# max_records_per_query: 500,
|
||||
# supported_formats: [],
|
||||
# advanced_queries_supported: false
|
||||
# }
|
||||
@metadata = nil
|
||||
|
||||
@user = user
|
||||
|
||||
@url = nil
|
||||
@ids_total = 0
|
||||
@ids_retrieved = 0
|
||||
@block_size = 0
|
||||
@current_stream_status = true
|
||||
@last_stream_status = true
|
||||
@ids = nil
|
||||
end
|
||||
|
||||
# Factory method
|
||||
# @param user ::User
|
||||
# @return CartoDB::Datasources::Url::ArcGIS
|
||||
def self.get_new(user)
|
||||
return new(user)
|
||||
end
|
||||
|
||||
# @return String
|
||||
def to_s
|
||||
"<CartoDB::Datasources::Url::ArcGis @url=#{@url} @metadata=#{@metadata} @ids_total=#{@ids_total}" +
|
||||
" @ids_retrieved=#{@ids_retrieved} current_block_size=#{block_size(update=false)}>"
|
||||
end
|
||||
|
||||
# If will provide a url to download the resource, or requires calling get_resource()
|
||||
# @return Bool
|
||||
def providers_download_url?
|
||||
false
|
||||
end
|
||||
|
||||
# Perform the listing and return results
|
||||
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
|
||||
# @return [ Hash ]
|
||||
def get_resources_list(filter=[])
|
||||
filter
|
||||
end
|
||||
|
||||
# Retrieves a resource and returns its contents
|
||||
# @param id string
|
||||
# @return mixed
|
||||
def get_resource(id)
|
||||
raise 'Not supported by this datasource'
|
||||
end
|
||||
|
||||
# Initial stream, to be used for container creation (table usually)
|
||||
# @param id string
|
||||
# @return String
|
||||
def initial_stream(id)
|
||||
sub_id = get_subresource_id(id)
|
||||
@url = sanitize_id(id, sub_id)
|
||||
|
||||
@ids = get_ids_list(@url)
|
||||
|
||||
@ids_total = @ids.length
|
||||
|
||||
first_item = get_by_ids(@url, [@ids.slice!(0)], @metadata[:fields])
|
||||
@ids_retrieved += 1
|
||||
|
||||
# Start optimistic
|
||||
@block_size = [MAX_BLOCK_SIZE, @metadata[:max_records_per_query]].min
|
||||
|
||||
::JSON.dump(first_item)
|
||||
end
|
||||
|
||||
# @param id string
|
||||
# @return String|nil Nil if no more items
|
||||
def stream_resource(id)
|
||||
return nil if @ids.empty?
|
||||
|
||||
retries = 0
|
||||
begin
|
||||
ids_block = @ids.slice!(0, [@ids.length, block_size].min)
|
||||
|
||||
puts "#{@ids_retrieved}/#{@ids_total} (#{ids_block.length})" if DEBUG
|
||||
|
||||
items = get_by_ids(@url, ids_block, @metadata[:fields])
|
||||
@last_stream_status = @current_stream_status
|
||||
@current_stream_status = true
|
||||
retries = 0
|
||||
sleep(BLOCK_SLEEP_TIME) unless BLOCK_SLEEP_TIME == 0
|
||||
rescue ExternalServiceError => exception
|
||||
if @block_size == MIN_BLOCK_SIZE && retries >= MAX_RETRIES
|
||||
if SKIP_FAILED_IDS
|
||||
items = []
|
||||
else
|
||||
raise exception
|
||||
end
|
||||
else
|
||||
@last_stream_status = @current_stream_status
|
||||
@current_stream_status = false
|
||||
# Add back, next pass will get fewer items
|
||||
@ids = ids_block + @ids
|
||||
|
||||
if @block_size == MIN_BLOCK_SIZE
|
||||
retries += 1
|
||||
sleep_time = SLEEP_REQUEST_TIME ** retries
|
||||
puts "Retry delay (#{sleep_time}s)" if DEBUG
|
||||
sleep(sleep_time)
|
||||
end
|
||||
|
||||
retry
|
||||
end
|
||||
end
|
||||
|
||||
@ids_retrieved += ids_block.length
|
||||
|
||||
items.length > 0 ? ::JSON.dump(items) : ''
|
||||
end
|
||||
|
||||
# @param id string
|
||||
# @return Hash
|
||||
# @throws DataDownloadError
|
||||
# @throws ResponseError
|
||||
# @throws InvalidServiceError
|
||||
# @throws ServiceDisabledError
|
||||
def get_resource_metadata(id)
|
||||
if is_multiresource?(id)
|
||||
@url = sanitize_id(id)
|
||||
{
|
||||
# Store original id, not the sanitized one
|
||||
id: id,
|
||||
subresources: get_layers_list(@url)
|
||||
}
|
||||
else
|
||||
sub_id = get_subresource_id(id)
|
||||
get_subresource_metadata(id, sub_id)
|
||||
end
|
||||
end
|
||||
|
||||
# Retrieves current filters. Unused as here there's no get_resources_list
|
||||
# @return {}
|
||||
def filter
|
||||
{}
|
||||
end
|
||||
|
||||
# Sets current filters. Unused as here there's no get_resources_list
|
||||
# @param filter_data {}
|
||||
def filter=(filter_data=[])
|
||||
filter_data
|
||||
end
|
||||
|
||||
# If this datasource accepts a data import instance
|
||||
# @return Boolean
|
||||
def persists_state_via_data_import?
|
||||
false
|
||||
end
|
||||
|
||||
# If true, a single resource id might return >1 subresources (each one spawning a table)
|
||||
# @param id String
|
||||
# @return Bool
|
||||
def multi_resource_import_supported?(id)
|
||||
is_multiresource?(id)
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def http_client
|
||||
@http_client ||= Carto::Http::Client.get('arcgis')
|
||||
end
|
||||
|
||||
# @param id String
|
||||
# @param subresource_id String
|
||||
# @return Hash
|
||||
# @throws DataDownloadError
|
||||
# @throws ResponseError
|
||||
# @throws InvalidServiceError
|
||||
def get_subresource_metadata(id, subresource_id)
|
||||
@url = sanitize_id(id, subresource_id)
|
||||
|
||||
response = http_client.get(METADATA_URL % [@url], http_options)
|
||||
validate_response(METADATA_URL % [@url], response)
|
||||
|
||||
# non-rails symbolize keys
|
||||
data = ::JSON.parse(response.body).inject({}){|memo,(k,v)| memo[k.to_sym] = v; memo}
|
||||
|
||||
raise ResponseError.new("Invalid layer type: '#{data[:type]}'") if data[:type] != VECTOR_LAYER_TYPE
|
||||
raise ResponseError.new("Missing data: 'fields'") if data[:fields].nil?
|
||||
|
||||
if data[:supportedQueryFormats].present?
|
||||
supported_formats = data.fetch(:supportedQueryFormats).gsub(' ', '').split(',')
|
||||
else
|
||||
supported_formats = []
|
||||
end
|
||||
|
||||
begin
|
||||
@metadata = {
|
||||
arcgis_version: data.fetch(:currentVersion),
|
||||
name: data.fetch(:name),
|
||||
description: data.fetch(:description, ''),
|
||||
type: data.fetch(:type),
|
||||
geometry_type: data.fetch(:geometryType),
|
||||
copyright: data.fetch(:copyrightText, ''),
|
||||
fields: data.fetch(:fields).try(:map) { |field|
|
||||
{
|
||||
name: field['name'],
|
||||
type: field['type']
|
||||
}
|
||||
},
|
||||
max_records_per_query: data.fetch(:maxRecordCount, 500),
|
||||
supported_formats: supported_formats,
|
||||
advanced_queries_supported: data.fetch(:supportsAdvancedQueries, false)
|
||||
}
|
||||
rescue => exception
|
||||
raise ResponseError.new("Missing data: #{exception.to_s} #{exception.backtrace}")
|
||||
end
|
||||
|
||||
raise InvalidServiceError.new("Unsupported ArcGIS version #{@metadata[:arcgis_Version]}, must be >= #{MINIMUM_SUPPORTED_VERSION}") \
|
||||
if @metadata[:arcgis_version] < MINIMUM_SUPPORTED_VERSION
|
||||
|
||||
{
|
||||
id: id,
|
||||
title: @metadata[:name],
|
||||
url: nil,
|
||||
service: DATASOURCE_NAME,
|
||||
checksum: nil,
|
||||
size: NO_CONTENT_SIZE_PROVIDED,
|
||||
filename: filename_from(@metadata[:name])
|
||||
}
|
||||
end
|
||||
|
||||
# Just detects if id is a full map or a specific layer
|
||||
# @param id String
|
||||
# @return Bool
|
||||
def is_multiresource?(id)
|
||||
unless id.rindex('?').nil?
|
||||
id = id.slice(0, id.rindex('?'))
|
||||
end
|
||||
(id =~ /\/([0-9]+\/|[0-9]+)$/).nil?
|
||||
end
|
||||
|
||||
def get_subresource_id(id)
|
||||
id.match(/([0-9]+\/|[0-9]+)$?/)[0]
|
||||
end
|
||||
|
||||
# @param id String
|
||||
# @param sub_id String|nil
|
||||
# @return String
|
||||
# @throws InvalidInputDataError
|
||||
def sanitize_id(id, sub_id=nil)
|
||||
# http://<host>/<site>/rest/services/<folder>/<serviceName>/<serviceType>/
|
||||
# <site> is almost always "arcgis" (according to official doc)
|
||||
unless id =~ ARCGIS_API_LIKE_URL_RE
|
||||
raise InvalidInputDataError.new("Url doesn't looks as from ArcGIS server")
|
||||
end
|
||||
|
||||
unless id.rindex('?').nil?
|
||||
id = id.slice(0, id.rindex('?'))
|
||||
end
|
||||
if is_multiresource?(id) && !sub_id.nil?
|
||||
id = (id.end_with?('/') ? id : id + '/') + sub_id
|
||||
end
|
||||
|
||||
id
|
||||
end
|
||||
|
||||
# @return Array [ { :id, :title} ]
|
||||
# @throws DataDownloadError
|
||||
# @throws ResponseError
|
||||
def get_layers_list(url)
|
||||
request_url = LAYERS_URL % [url]
|
||||
response = http_client.get(request_url, http_options)
|
||||
validate_response(request_url, response)
|
||||
|
||||
begin
|
||||
data = ::JSON.parse(response.body).fetch('layers')
|
||||
rescue => exception
|
||||
raise ResponseError.new("Missing data: #{exception.to_s} #{request_url} #{exception.backtrace}")
|
||||
end
|
||||
|
||||
# We only support vector layers (not raster layers)
|
||||
data = data.reject { |layer| layer['type'] != VECTOR_LAYER_TYPE }
|
||||
|
||||
raise ResponseError.new("Empty layers list #{request_url}") if data.length == 0
|
||||
|
||||
begin
|
||||
data.collect { |item|
|
||||
{
|
||||
# Leave prepared all child urls
|
||||
id: (url.end_with?('/') ? url : url + '/') + item.fetch('id').to_s,
|
||||
title: item.fetch('name')
|
||||
}
|
||||
}
|
||||
rescue => exception
|
||||
raise ResponseError.new("Missing data: #{exception.to_s} #{request_url} #{exception.backtrace}")
|
||||
end
|
||||
end
|
||||
|
||||
# NOTE: Assumes url is valid
|
||||
# NOTE: Returned ids are sorted so they can be chunked into blocks to
|
||||
# be requested by range queries: `(OBJECTID >= ... AND OBJECTID <= ... )`
|
||||
# @param url String
|
||||
# @return Array
|
||||
# @throws DataDownloadError
|
||||
# @throws ResponseError
|
||||
def get_ids_list(url)
|
||||
request_url = FEATURE_IDS_URL % [url]
|
||||
response = http_client.get(request_url, http_options)
|
||||
validate_response(request_url, response)
|
||||
|
||||
begin
|
||||
data = ::JSON.parse(response.body).fetch('objectIds').sort
|
||||
rescue => exception
|
||||
raise ResponseError.new("Missing data: #{exception.to_s} #{request_url} #{exception.backtrace}")
|
||||
end
|
||||
|
||||
raise ResponseError.new("Empty ids list #{request_url}") if data.length == 0
|
||||
|
||||
data
|
||||
end
|
||||
|
||||
# NOTE: Assumes url is valid
|
||||
# @param url String
|
||||
# @param ids Array
|
||||
# @param fields Array
|
||||
# @return Array [ Hash ] (non-symbolized keys)
|
||||
# @throws InvalidInputDataError
|
||||
# @throws DataDownloadError
|
||||
# @throws ExternalServiceError
|
||||
def get_by_ids(url, ids, fields)
|
||||
raise InvalidInputDataError.new("'ids' empty or invalid") if (ids.nil? || ids.length == 0)
|
||||
raise InvalidInputDataError.new("'fields' empty or invalid") if (fields.nil? || fields.length == 0)
|
||||
|
||||
oid_field = fields.find { |field| field[:type] == OID_FIELD_TYPE }
|
||||
|
||||
if ids.length == 1
|
||||
ids_field = { objectIds: ids.first }
|
||||
else
|
||||
if oid_field
|
||||
# Note that ids is sorted
|
||||
ids_field = { where: "#{oid_field[:name]} >=#{ids.first} AND #{oid_field[:name]} <=#{ids.last}" }
|
||||
else
|
||||
# This could be innefficient with large number of ids, but it is limited to MAX_BLOCK_SIZE
|
||||
ids_field = { objectIds: ids.join(',') }
|
||||
end
|
||||
end
|
||||
|
||||
prepared_fields = fields.map { |field| "#{field[:name]}" }.join(',')
|
||||
|
||||
prepared_url = FEATURE_DATA_POST_URL % [url]
|
||||
# @see http://resources.arcgis.com/en/help/arcgis-rest-api/index.html#/Query_Map_Service_Layer/02r3000000p1000000/
|
||||
params_data = {
|
||||
outFields: prepared_fields,
|
||||
outSR: 4326,
|
||||
f: 'json'
|
||||
}
|
||||
|
||||
params_data.merge! ids_field
|
||||
|
||||
puts "#{prepared_url} (POST) Params:#{params_data}" if DEBUG
|
||||
response = http_client.post(prepared_url, http_options(params_data, :post))
|
||||
|
||||
# Timeout connecting to ArcGIS
|
||||
if response.code == 0
|
||||
raise ExternalServiceError.new("TIMEOUT: #{prepared_url} : #{response.body} #{self.to_s}")
|
||||
end
|
||||
if response.code != 200
|
||||
raise DataDownloadError.new("ERROR: #{prepared_url} POST " +
|
||||
"#{params_data} (#{response.code}) : #{response.body} #{self.to_s}")
|
||||
end
|
||||
if response.code == 400 && !response.return_message.nil? \
|
||||
&& response.return_message.downcase.include?('operation is not supported')
|
||||
raise UnsupportedOperationError.new("#{request_url} (#{response.code}) : #{response.body}") \
|
||||
end
|
||||
|
||||
begin
|
||||
body = ::JSON.parse(response.body)
|
||||
success = true
|
||||
rescue JSON::ParserError
|
||||
success = false
|
||||
end
|
||||
|
||||
unless success
|
||||
begin
|
||||
# HACK: JSON spec does not cover Infinity
|
||||
body = ::JSON.parse(response.body.gsub(':INF,', ':"Infinity",'))
|
||||
rescue JSON::ParserError
|
||||
raise ResponseError.new("JSON parsing error. URL: #{prepared_url} #{to_s}")
|
||||
end
|
||||
end
|
||||
|
||||
# Arcgis error
|
||||
raise ExternalServiceError.new("#{prepared_url} : #{response.body}") if body.include?('error')
|
||||
|
||||
begin
|
||||
retrieved_items = body.fetch('features')
|
||||
return [] if retrieved_items.nil? || retrieved_items.empty?
|
||||
retrieved_fields = body.fetch('fields')
|
||||
geometry_type = body.fetch('geometryType')
|
||||
spatial_reference = body.fetch('spatialReference')
|
||||
rescue => exception
|
||||
raise ResponseError.new("Missing data: #{exception.to_s} #{prepared_url} #{exception.backtrace}")
|
||||
end
|
||||
raise ResponseError.new("'fields' empty or invalid #{prepared_url}") \
|
||||
if (retrieved_fields.nil? || retrieved_fields.length == 0)
|
||||
raise ResponseError.new("'features' empty or invalid #{prepared_url}") \
|
||||
if (retrieved_items.nil? || !retrieved_items.kind_of?(Array))
|
||||
|
||||
# Fields can be optional, cannot be enforced to always be present
|
||||
desired_fields = fields.map { |field| field[:name] }
|
||||
|
||||
{
|
||||
geometryType: geometry_type,
|
||||
spatialReference: spatial_reference,
|
||||
fields: retrieved_fields,
|
||||
features: retrieved_items.collect { |item|
|
||||
{
|
||||
'attributes' => item['attributes'].select{ |k, v| desired_fields.include?(k) },
|
||||
'geometry' => item['geometry']
|
||||
}
|
||||
}
|
||||
}
|
||||
end
|
||||
|
||||
# By default, will update the block size, incrementing or decrementing it according to stream operation results
|
||||
# Block size only gets incremented after 2 successful streams to avoid scenario of:
|
||||
# X items -> FAIL
|
||||
# X/2 items -> PASS
|
||||
# X items -> FAIL (again, because erroring item was at second half of X)
|
||||
def block_size(update=true)
|
||||
if update
|
||||
if @current_stream_status && @last_stream_status && @block_size < MAX_BLOCK_SIZE
|
||||
@block_size = [@block_size * BLOCK_FACTOR, MAX_BLOCK_SIZE].min
|
||||
end
|
||||
if !@current_stream_status && @block_size > MIN_BLOCK_SIZE
|
||||
@block_size = [[(@block_size / BLOCK_FACTOR).floor, 1].max, MAX_BLOCK_SIZE].min
|
||||
end
|
||||
@block_size = [@block_size, @metadata[:max_records_per_query]].min
|
||||
end
|
||||
@block_size
|
||||
end
|
||||
|
||||
def http_options(params={}, method=:get)
|
||||
{
|
||||
method: method,
|
||||
params: method == :get ? params : {},
|
||||
body: method == :post ? params : {},
|
||||
followlocation: true,
|
||||
ssl_verifypeer: false,
|
||||
accept_encoding: 'gzip',
|
||||
headers: { 'Accept-Charset' => 'utf-8' },
|
||||
ssl_verifyhost: 0,
|
||||
nosignal: true,
|
||||
connecttimeout: HTTP_CONNECTION_TIMEOUT,
|
||||
timeout: HTTP_TIMEOUT
|
||||
}
|
||||
end
|
||||
|
||||
def filename_from(feature_name)
|
||||
feature_name.gsub(/[^\w]/, '_').downcase + '.json'
|
||||
end
|
||||
|
||||
def validate_response(request_url, response)
|
||||
raise ExternalServiceTimeoutError.new("TIMEOUT: #{request_url} : #{response.return_message}") \
|
||||
if response.timed_out? || (response.code.zero? && !response.return_message.nil? \
|
||||
&& response.return_message.downcase.include?('timeout'))
|
||||
|
||||
raise UnsupportedOperationError.new("#{request_url} (#{response.code}) : #{response.body}") \
|
||||
if response.code == 400 && !response.return_message.nil? \
|
||||
&& response.return_message.downcase.include?('operation is not supported')
|
||||
|
||||
raise DataDownloadError.new("#{request_url} (#{response.code}) : #{response.body}") \
|
||||
if response.code != 200
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
class URLTooLargeError < StandardError
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,543 @@
|
||||
require_relative '../../../../../lib/carto/http/client'
|
||||
|
||||
module CartoDB
|
||||
module Datasources
|
||||
module Url
|
||||
# BoxAPI module is a replacement for Boxr, which requires Ruby >= 2.
|
||||
# Most of this code has been extracted from Boxr.
|
||||
# This should be migrated when we upgrade to Ruby 2.
|
||||
module BoxAPI
|
||||
class ExpiredTokenError < StandardError; end
|
||||
|
||||
def self.oauth_url(state, options = {})
|
||||
host = options.fetch(:host, "app.box.com")
|
||||
response_type = options.fetch(:response_type, "code")
|
||||
scope = options[:scope]
|
||||
folder_id = options[:folder_id]
|
||||
client_id = options[:client_id]
|
||||
|
||||
template = Addressable::Template.new("https://{host}/api/oauth2/authorize{?query*}")
|
||||
|
||||
query = { "response_type" => "#{response_type}", "state" => "#{state}", "client_id" => "#{client_id}" }
|
||||
query["scope"] = "#{scope}" unless scope.nil?
|
||||
query["folder_id"] = "#{folder_id}" unless folder_id.nil?
|
||||
|
||||
template.expand("host" => "#{host}", "query" => query)
|
||||
end
|
||||
|
||||
def self.get_tokens(code, options = {})
|
||||
grant_type = options[:grant_type]
|
||||
assertion = options[:assertion]
|
||||
scope = options[:scope]
|
||||
username = options[:username]
|
||||
client_id = options[:client_id]
|
||||
client_secret = options[:client_secret]
|
||||
|
||||
uri = "https://api.box.com/oauth2/token"
|
||||
body = "grant_type=#{grant_type}&client_id=#{client_id}&client_secret=#{client_secret}"
|
||||
body = body + "&code=#{code}" unless code.nil?
|
||||
body = body + "&scope=#{scope}" unless scope.nil?
|
||||
body = body + "&username=#{username}" unless username.nil?
|
||||
body = body + "&assertion=#{assertion}" unless assertion.nil?
|
||||
|
||||
auth_post(uri, body)
|
||||
end
|
||||
|
||||
def self.refresh_tokens(refresh_token, options = {})
|
||||
client_id = options[:client_id]
|
||||
client_secret = options[:client_secret]
|
||||
|
||||
uri = "https://api.box.com/oauth2/token"
|
||||
body = "grant_type=refresh_token&refresh_token=#{refresh_token}&client_id=#{client_id}&client_secret=#{client_secret}"
|
||||
|
||||
auth_post(uri, body)
|
||||
end
|
||||
|
||||
def self.auth_post(uri, body, json_body = true)
|
||||
uri = Addressable::URI.encode(uri)
|
||||
|
||||
res = post(uri, body: body)
|
||||
|
||||
if res.response_code == 200
|
||||
json_body ? JSON.parse(res.response_body) : res.response_body
|
||||
else
|
||||
handle_error_response(res)
|
||||
end
|
||||
end
|
||||
|
||||
def self.handle_error_response(res)
|
||||
body_json = JSON.parse(res.response_body)
|
||||
|
||||
if body_json['error'] == 'invalid_grant'
|
||||
raise ExpiredTokenError.new(body_json.fetch('error_description', 'Expired token'))
|
||||
else
|
||||
raise "Box Error status: #{res.response_code}, body: #{res.response_body}, headers: #{res.response_headers}"
|
||||
end
|
||||
rescue => e
|
||||
CartoDB.notify_exception(e, self: inspect, response: res.inspect)
|
||||
raise e
|
||||
end
|
||||
|
||||
def self.post(uri, options = {})
|
||||
http_client = Carto::Http::Client.get('box',
|
||||
connecttimeout: 60,
|
||||
timeout: 600
|
||||
)
|
||||
http_client.post(uri.to_s, options)
|
||||
end
|
||||
|
||||
def self.get(uri, options = {})
|
||||
query = options[:query]
|
||||
header = options[:header]
|
||||
follow_redirect = options[:follow_redirect]
|
||||
|
||||
http_client = Carto::Http::Client.get('box',
|
||||
connecttimeout: 60,
|
||||
timeout: 600)
|
||||
response = http_client.get(uri.to_s, headers: header, followlocation: follow_redirect, params: query)
|
||||
response
|
||||
end
|
||||
end
|
||||
|
||||
module BoxAPI
|
||||
class Client
|
||||
|
||||
API_URI = "https://api.box.com/2.0"
|
||||
SEARCH_URI = "#{API_URI}/search"
|
||||
FILES_URI = "#{API_URI}/files"
|
||||
|
||||
def initialize(access_token, options = {})
|
||||
client_id = options[:client_id]
|
||||
client_secret = options[:client_secret]
|
||||
|
||||
@access_token = access_token
|
||||
raise "Access token cannot be nil" if @access_token.nil?
|
||||
@client_id = client_id
|
||||
@client_secret = client_secret
|
||||
end
|
||||
|
||||
def search(query, options = {})
|
||||
scope = options[:scope]
|
||||
file_extensions = options[:file_extensions]
|
||||
created_at_range = options[:created_at_range]
|
||||
updated_at_range = options[:updated_at_range]
|
||||
size_range = options[:size_range]
|
||||
owner_user_ids = options[:owner_user_ids]
|
||||
ancestor_folder_ids = options[:ancestor_folder_ids]
|
||||
content_types = options[:content_types]
|
||||
type = options[:type]
|
||||
limit = options.fetch(:limit, 30)
|
||||
offset = options.fetch(:offset, 0)
|
||||
|
||||
query = { query: query }
|
||||
query[:scope] = scope unless scope.nil?
|
||||
query[:file_extensions] = file_extensions unless file_extensions.nil?
|
||||
query[:created_at_range] = created_at_range unless created_at_range.nil?
|
||||
query[:updated_at_range] = updated_at_range unless updated_at_range.nil?
|
||||
query[:size_range] = size_range unless size_range.nil?
|
||||
query[:owner_user_ids] = owner_user_ids unless owner_user_ids.nil?
|
||||
query[:ancestor_folder_ids] = ancestor_folder_ids unless ancestor_folder_ids.nil?
|
||||
query[:content_types] = content_types unless content_types.nil?
|
||||
query[:type] = type unless type.nil?
|
||||
query[:limit] = limit unless limit.nil?
|
||||
query[:offset] = offset unless offset.nil?
|
||||
|
||||
results, _response = get(SEARCH_URI, query: query)
|
||||
results['entries']
|
||||
end
|
||||
|
||||
def download_url(file, options = {})
|
||||
version = options[:version]
|
||||
|
||||
download_file(file, version: version, follow_redirect: false)
|
||||
end
|
||||
|
||||
def download_file(file, options = {})
|
||||
version = options[:version]
|
||||
follow_redirect = options.fetch(:follow_redirect, true)
|
||||
|
||||
file_id = ensure_id(file)
|
||||
begin
|
||||
uri = "#{FILES_URI}/#{file_id}/content"
|
||||
query = {}
|
||||
query[:version] = version unless version.nil?
|
||||
|
||||
# Boxr didn't have 200
|
||||
_body_json, response = get(uri, query: query, success_codes: [302, 202, 200], follow_redirect: false, process_response: false)
|
||||
|
||||
if response.response_code == 302
|
||||
location = response.header['Location'][0]
|
||||
|
||||
if follow_redirect
|
||||
file, response = get(location, process_response: false)
|
||||
else
|
||||
return location # simply return the url
|
||||
end
|
||||
elsif response.response_code == 202
|
||||
retry_after_seconds = response.header['Retry-After'][0]
|
||||
sleep retry_after_seconds.to_i
|
||||
elsif response.response_code == 200
|
||||
file = response.response_body
|
||||
end
|
||||
end until file
|
||||
|
||||
file
|
||||
end
|
||||
|
||||
FOLDER_AND_FILE_FIELDS = [:type, :id, :sequence_id, :etag, :name, :created_at, :modified_at, :description,
|
||||
:size, :path_collection, :created_by, :modified_by, :trashed_at, :purged_at,
|
||||
:content_created_at, :content_modified_at, :owned_by, :shared_link,
|
||||
:folder_upload_email,
|
||||
:parent, :item_status, :item_collection, :sync_state, :has_collaborations,
|
||||
:permissions, :tags,
|
||||
:sha1, :shared_link, :version_number, :comment_count, :lock, :extension,
|
||||
:is_package,
|
||||
:expiring_embed_link, :can_non_owners_invite]
|
||||
FOLDER_AND_FILE_FIELDS_QUERY = FOLDER_AND_FILE_FIELDS.join(',')
|
||||
|
||||
def file_from_id(file_id, fields = [])
|
||||
file_id = ensure_id(file_id)
|
||||
uri = "#{FILES_URI}/#{file_id}"
|
||||
query = build_fields_query(fields, FOLDER_AND_FILE_FIELDS_QUERY)
|
||||
file, _response = get(uri, query: query)
|
||||
file
|
||||
end
|
||||
|
||||
# Required for all providers
|
||||
DATASOURCE_NAME = 'box'
|
||||
|
||||
def revoke_tokens(token)
|
||||
uri = "https://api.box.com/oauth2/revoke"
|
||||
body = "client_id=#{@client_id}&client_secret=#{@client_secret}&token=#{token}"
|
||||
|
||||
BoxAPI::auth_post(uri, body, false)
|
||||
rescue => ex
|
||||
raise AuthError.new("revoke_token: #{ex.message}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def build_fields_query(fields, all_fields_query)
|
||||
if fields == :all
|
||||
{ fields: all_fields_query }
|
||||
elsif fields.is_a?(Array) && fields.length > 0
|
||||
{ fields: fields.join(',') }
|
||||
else
|
||||
{}
|
||||
end
|
||||
end
|
||||
|
||||
def ensure_id(item)
|
||||
return item if item.class == String || item.class == Integer || item.nil?
|
||||
return item.id if item.respond_to?(:id)
|
||||
return item['id'] if item.class == Hash
|
||||
raise "Expecting an id of class String or Integer, or object that responds to :id"
|
||||
end
|
||||
|
||||
def get(uri, options = {})
|
||||
query = options[:query]
|
||||
success_codes = options.fetch(:success_codes, [200])
|
||||
process_response = options.fetch(:process_response, true)
|
||||
if_match = options[:if_match]
|
||||
box_api_header = options[:box_api_header]
|
||||
follow_redirect = options.fetch(follow_redirect, true)
|
||||
|
||||
headers = standard_headers
|
||||
headers['If-Match'] = if_match unless if_match.nil?
|
||||
headers['BoxApi'] = box_api_header unless box_api_header.nil?
|
||||
|
||||
res = BoxAPI::get(uri, query: query, header: headers, follow_redirect: follow_redirect)
|
||||
|
||||
check_response_status(res, success_codes)
|
||||
|
||||
if process_response
|
||||
return JSON.parse(res.response_body)
|
||||
else
|
||||
return res.response_body, res
|
||||
end
|
||||
end
|
||||
|
||||
def check_response_status(res, success_codes)
|
||||
unless success_codes.include?(res.response_code)
|
||||
raise "BoxError status: #{res.response_code}, body: #{res.response_body}, header: #{res.response_headers}"
|
||||
end
|
||||
end
|
||||
|
||||
def standard_headers
|
||||
{ "Authorization" => "Bearer #{@access_token}" }
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
|
||||
class Box < BaseOAuth
|
||||
|
||||
# Required for all providers
|
||||
DATASOURCE_NAME = 'box'
|
||||
|
||||
# Constructor (hidden)
|
||||
# @param config
|
||||
# [
|
||||
# 'application_name'
|
||||
# 'client_id'
|
||||
# 'client_secret'
|
||||
# ]
|
||||
# @param user ::User
|
||||
# @throws UninitializedError
|
||||
# @throws MissingConfigurationError
|
||||
def initialize(config, user)
|
||||
super(config, user, %w{ application_name client_id client_secret box_host }, DATASOURCE_NAME)
|
||||
|
||||
raise UninitializedError.new('missing user instance', DATASOURCE_NAME) if user.nil?
|
||||
|
||||
@access_token = nil
|
||||
@refresh_token = nil
|
||||
end
|
||||
|
||||
# Factory method
|
||||
# @param config {}
|
||||
# @param user ::User
|
||||
# @return CartoDB::Datasources::Url::GDrive
|
||||
def self.get_new(config, user)
|
||||
new(config, user)
|
||||
end
|
||||
|
||||
# If will provide a url to download the resource, or requires calling get_resource()
|
||||
# @return bool
|
||||
def providers_download_url?
|
||||
false
|
||||
end
|
||||
|
||||
# Return the url to be displayed or sent the user to to authenticate and get authorization code.
|
||||
# Older implementations had a use_callback_flow parameter that became deprecated. Not implemented.
|
||||
# @return string | nil
|
||||
def get_auth_url
|
||||
service_name = service_name_for_user(DATASOURCE_NAME, @user)
|
||||
state = CALLBACK_STATE_DATA_PLACEHOLDER.sub('user', @user.username).sub('service', service_name)
|
||||
BoxAPI::oauth_url(state,
|
||||
host: config['box_host'],
|
||||
response_type: "code",
|
||||
scope: nil,
|
||||
folder_id: nil,
|
||||
client_id: config['client_id']).to_s
|
||||
end
|
||||
|
||||
# Validates authorization code, sets access and refresh tokens for current instance and stores (refresh) token
|
||||
# @param auth_code : string
|
||||
# @return string : Refresh token
|
||||
# Older implementations had a use_callback_flow parameter that became deprecated. Not implemented.
|
||||
# @throws AuthError
|
||||
def validate_auth_code(auth_code)
|
||||
set_tokens(get_tokens(auth_code))
|
||||
@refresh_token
|
||||
end
|
||||
|
||||
# Validates the authorization callback
|
||||
# @param params : mixed
|
||||
def validate_callback(params)
|
||||
if params[:error].present?
|
||||
raise AuthError.new("validate_callback: #{params[:error]}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
if params[:code]
|
||||
validate_auth_code(params[:code])
|
||||
else
|
||||
raise AuthError.new('validate_callback: Missing authorization code', DATASOURCE_NAME)
|
||||
end
|
||||
end
|
||||
|
||||
# Store (refresh) token. If it's not valid both access_token and refresh_token will be nil.
|
||||
# Triggers generation of a valid access token for the lifetime of this instance
|
||||
# @param token string
|
||||
# @throws AuthError
|
||||
def token=(token)
|
||||
set_tokens(get_fresh_tokens(token))
|
||||
rescue CartoDB::Datasources::Url::BoxAPI::ExpiredTokenError => e
|
||||
CartoDB.notify_exception(e, self: inspect, token: token)
|
||||
set_tokens('access_token' => nil, 'refresh_token' => nil)
|
||||
end
|
||||
|
||||
# Retrieve (refresh) token
|
||||
# @return string | nil
|
||||
def token
|
||||
@refresh_token
|
||||
end
|
||||
|
||||
# Perform the listing and return results
|
||||
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
|
||||
# @return [ { :id, :title, :url, :service } ]
|
||||
# @throws TokenExpiredOrInvalidError
|
||||
# @throws DataDownloadError
|
||||
def get_resources_list(filter = [])
|
||||
self.filter = filter
|
||||
|
||||
# Box doesn't have a way to "retrieve everything" but it supports whitespaces for multiple search terms
|
||||
result = client.search(supported_extensions.join(' '),
|
||||
scope: nil,
|
||||
file_extensions: nil,
|
||||
created_at_range: nil,
|
||||
updated_at_range: nil,
|
||||
size_range: nil,
|
||||
owner_user_ids: nil,
|
||||
ancestor_folder_ids: nil,
|
||||
content_types: nil,
|
||||
type: nil,
|
||||
limit: 200,
|
||||
offset: 0)
|
||||
|
||||
result = result.map { |i| format_item_data(i) }.sort { |x, y| y[:updated_at] <=> x[:updated_at] }
|
||||
|
||||
unless @formats.nil? || @formats.empty?
|
||||
result = result.select { |item| item[:filename] =~ /.*(#{@formats.join(')|(')})$/i }
|
||||
end
|
||||
|
||||
result
|
||||
end
|
||||
|
||||
# Retrieves a resource and returns its contents
|
||||
# @param id string
|
||||
# @return mixed
|
||||
# @throws TokenExpiredOrInvalidError
|
||||
# @throws DataDownloadError
|
||||
def get_resource(id)
|
||||
file = client.file_from_id(id)
|
||||
client.download_file(file)
|
||||
end
|
||||
|
||||
# @param id string
|
||||
# @return Hash
|
||||
# @throws TokenExpiredOrInvalidError
|
||||
# @throws DataDownloadError
|
||||
# @throws NotFoundDownloadError
|
||||
def get_resource_metadata(id)
|
||||
result = client.file_from_id(id)
|
||||
|
||||
if result.nil?
|
||||
message = "Retrieving file #{id} metadata: #{result.inspect}, should stop syncing"
|
||||
raise NotFoundDownloadError.new(message, DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
if result['item_status'] != 'active'
|
||||
raise DataDownloadError.new("Retrieving file #{id} metadata: #{result.inspect}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
format_item_data(result)
|
||||
end
|
||||
|
||||
# Retrieves current filters
|
||||
# @return {}
|
||||
def filter
|
||||
@formats
|
||||
end
|
||||
|
||||
# Sets current filters
|
||||
# @param filter_data {}
|
||||
def filter=(filter_data = [])
|
||||
@formats = filter_data
|
||||
end
|
||||
|
||||
# Just return datasource name
|
||||
# @return string
|
||||
def to_s
|
||||
DATASOURCE_NAME
|
||||
end
|
||||
|
||||
# If this datasource accepts a data import instance
|
||||
# @return Boolean
|
||||
def persists_state_via_data_import?
|
||||
false
|
||||
end
|
||||
|
||||
# Stores the data import item instance to use/manipulate it
|
||||
# @param value DataImport
|
||||
# Not implemented
|
||||
def data_import_item=(_value)
|
||||
nil
|
||||
end
|
||||
|
||||
# Checks if token is still valid or has been revoked
|
||||
# @return bool
|
||||
# @throws AuthError
|
||||
def token_valid?
|
||||
raise 'invalid_token' unless token
|
||||
|
||||
# Any call would do, we just want to see if communicates or refuses the token
|
||||
result = client.search('test search')
|
||||
!result.nil?
|
||||
rescue => e
|
||||
if e.message =~ /invalid_token/
|
||||
CartoDB.notify_debug('Box invalid_token', self: inspect)
|
||||
false
|
||||
else
|
||||
CartoDB.notify_exception(e, self: inspect)
|
||||
raise e
|
||||
end
|
||||
end
|
||||
|
||||
# Revokes current set token
|
||||
def revoke_token
|
||||
client.revoke_tokens(token)
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def set_tokens(tokens)
|
||||
@access_token = tokens['access_token']
|
||||
@refresh_token = tokens['refresh_token']
|
||||
end
|
||||
|
||||
def client
|
||||
@client ||= get_client
|
||||
end
|
||||
|
||||
def get_client
|
||||
BoxAPI::Client.new(@access_token,
|
||||
client_id: config['client_id'],
|
||||
client_secret: config['client_secret'])
|
||||
end
|
||||
|
||||
def get_tokens(code)
|
||||
BoxAPI::get_tokens(code,
|
||||
grant_type: "authorization_code",
|
||||
assertion: nil,
|
||||
scope: nil,
|
||||
username: nil,
|
||||
client_id: config['client_id'],
|
||||
client_secret: config['client_secret'])
|
||||
end
|
||||
|
||||
def get_fresh_tokens(refresh_token)
|
||||
tokens = BoxAPI::refresh_tokens(refresh_token,
|
||||
client_id: config['client_id'],
|
||||
client_secret: config['client_secret'])
|
||||
# Box refresh tokens can only be used once
|
||||
update_user_oauth(tokens['refresh_token'])
|
||||
tokens
|
||||
end
|
||||
|
||||
def update_user_oauth(refresh_token)
|
||||
carto_user = Carto::User.find(@user.id)
|
||||
oauth = carto_user.oauth_for_service('box')
|
||||
oauth.token = refresh_token
|
||||
oauth.save
|
||||
end
|
||||
|
||||
# Formats all data to comply with our desired format
|
||||
# @param item_data Hash : Single item returned from GDrive API
|
||||
# @return { :id, :title, :url, :service, :checksum, :size, :filename, :updated_at }
|
||||
def format_item_data(item_data)
|
||||
{
|
||||
id: item_data['id'],
|
||||
title: item_data['name'],
|
||||
service: DATASOURCE_NAME,
|
||||
checksum: checksum_of(item_data.fetch('modified_at')),
|
||||
filename: item_data['name'],
|
||||
size: item_data['size'].to_i,
|
||||
updated_at: DateTime.rfc3339(item_data['content_modified_at'])
|
||||
}
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,281 @@
|
||||
require 'dropbox_api'
|
||||
require_relative '../base_oauth'
|
||||
require_relative '../../../../../lib/dropbox_api/endpoints/auth/token/revoke'
|
||||
|
||||
module CartoDB
|
||||
module Datasources
|
||||
module Url
|
||||
# In order to test Dropbox in local, do the following:
|
||||
# 1. In Dropbox, change OAuth2 configuration, adding this (replace username and API key as needed):
|
||||
# http://localhost:3000/u/juanignaciosl/api/v1/imports/service/dropbox/oauth_callback/?api_key=3312b39c6360862e13217a8aec540e57367f4a4b
|
||||
# 2. Configure it in app_config.yml:
|
||||
# dropbox:
|
||||
# app_key: '528omteaww7fj86'
|
||||
# app_secret: 'rhx2ovpuni266ra'
|
||||
# callback_url: 'http://localhost:3000/u/juanignaciosl/api/v1/imports/service/dropbox/oauth_callback/?api_key=3312b39c6360862e13217a8aec540e57367f4a4b'
|
||||
# This obviously will work for a single user.
|
||||
class Dropbox < BaseOAuth
|
||||
|
||||
# Required for all datasources
|
||||
DATASOURCE_NAME = 'dropbox'
|
||||
|
||||
# Specific of this datasource
|
||||
FORMATS_TO_SEARCH_QUERIES = {
|
||||
FORMAT_CSV => %W( .csv ),
|
||||
FORMAT_EXCEL => %W( .xls .xlsx ),
|
||||
FORMAT_GPX => %W( .gpx ),
|
||||
FORMAT_KML => %W( .kml ),
|
||||
FORMAT_COMPRESSED => %W( .zip )
|
||||
}
|
||||
|
||||
# Constructor
|
||||
# @param config Array
|
||||
# [
|
||||
# 'app_key'
|
||||
# 'app_secret'
|
||||
# 'callback_url'
|
||||
# ]
|
||||
# @param user ::User
|
||||
# @throws UninitializedError
|
||||
# @throws MissingConfigurationError
|
||||
def initialize(config, user)
|
||||
super(config, user, %w{ app_key app_secret callback_url }, DATASOURCE_NAME)
|
||||
|
||||
@user = user
|
||||
@app_key = config.fetch('app_key')
|
||||
@app_secret = config.fetch('app_secret')
|
||||
@callback_url = config.fetch('callback_url')
|
||||
|
||||
self.filter = []
|
||||
@access_token = nil
|
||||
@auth_flow = nil
|
||||
@client = nil
|
||||
end
|
||||
|
||||
# Factory method
|
||||
# @param config : {}
|
||||
# @param user : ::User
|
||||
# @return CartoDB::Datasources::Url::Dropbox
|
||||
def self.get_new(config, user)
|
||||
return new(config, user)
|
||||
end
|
||||
|
||||
# If will provide a url to download the resource, or requires calling get_resource()
|
||||
# @return bool
|
||||
def providers_download_url?
|
||||
false
|
||||
end
|
||||
|
||||
# Return the url to be displayed or sent the user to to authenticate and get authorization code
|
||||
# Older implementations had a use_callback_flow parameter that became deprecated. Not implemented.
|
||||
# @throws AuthError
|
||||
def get_auth_url
|
||||
authenticator.authorize_url redirect_uri: @callback_url, state: state
|
||||
rescue => ex
|
||||
raise AuthError.new("get_auth_url(#{use_callback_flow}): #{ex.message}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
# Validates the authorization callback
|
||||
# @param params : mixed
|
||||
def validate_callback(params)
|
||||
raise "state doesn't match" unless params[:state] == state
|
||||
auth_bearer = authenticator.get_token(params[:code], redirect_uri: @callback_url)
|
||||
@access_token = auth_bearer.token
|
||||
|
||||
@client = DropboxApi::Client.new(@access_token)
|
||||
@access_token
|
||||
rescue => ex
|
||||
raise AuthError.new("validate_callback(#{params.inspect}): #{ex.message}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
# Set the token
|
||||
# @param token string
|
||||
# @throws TokenExpiredOrInvalidError
|
||||
# @throws AuthError
|
||||
def token=(token)
|
||||
@access_token = token
|
||||
@client = DropboxApi::Client.new(@access_token)
|
||||
rescue => ex
|
||||
handle_error(ex, "token= : #{ex.message}")
|
||||
end
|
||||
|
||||
# Retrieve set token
|
||||
# @return string | nil
|
||||
def token
|
||||
@access_token
|
||||
end
|
||||
|
||||
# Perform the listing and return results
|
||||
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
|
||||
# @return [ { :id, :title, :url, :service } ]
|
||||
# @throws TokenExpiredOrInvalidError
|
||||
# @throws AuthError
|
||||
# @throws DataDownloadError
|
||||
def get_resources_list(filter=[])
|
||||
all_results = []
|
||||
self.filter = filter
|
||||
|
||||
@formats.each do |search_query|
|
||||
start = 0
|
||||
loop do
|
||||
response = @client.search(search_query, '', max_results: SEARCH_BATCH_SIZE, start: start)
|
||||
response.matches.select { |item| item.resource.is_a?(DropboxApi::Metadata::File) }.each do |item|
|
||||
all_results.push(format_item_data(item.resource))
|
||||
end
|
||||
break unless response.has_more?
|
||||
start += SEARCH_BATCH_SIZE
|
||||
end
|
||||
end
|
||||
all_results
|
||||
rescue => ex
|
||||
handle_error(ex, "get_resources_list(): #{ex.message}")
|
||||
end
|
||||
|
||||
# Retrieves a resource and returns its contents
|
||||
# @param id string
|
||||
# @return mixed
|
||||
# @throws TokenExpiredOrInvalidError
|
||||
# @throws AuthError
|
||||
# @throws DataDownloadError
|
||||
def get_resource(id)
|
||||
file_contents = ''
|
||||
@client.download(id) do |chunk|
|
||||
file_contents << chunk
|
||||
end
|
||||
file_contents
|
||||
rescue => ex
|
||||
handle_error(ex, "get_resource() #{id}: #{ex.message}")
|
||||
end
|
||||
|
||||
# @param id string
|
||||
# @return Hash
|
||||
# @throws TokenExpiredOrInvalidError
|
||||
# @throws AuthError
|
||||
# @throws DataDownloadError
|
||||
def get_resource_metadata(id)
|
||||
raise DropboxPermissionError.new('No Dropbox client', DATASOURCE_NAME) unless @client.present?
|
||||
|
||||
response = @client.get_metadata(id)
|
||||
item_data = format_item_data(response)
|
||||
|
||||
item_data.to_hash
|
||||
rescue => ex
|
||||
handle_error(ex, "get_resource_metadata() #{id}: #{ex.message}")
|
||||
end
|
||||
|
||||
# Retrieves current filters
|
||||
# @return {}
|
||||
def filter
|
||||
@formats
|
||||
end
|
||||
|
||||
# Sets current filters
|
||||
# @param filter_data {}
|
||||
def filter=(filter_data=[])
|
||||
@formats = []
|
||||
FORMATS_TO_SEARCH_QUERIES.each do |id, queries|
|
||||
if filter_data.empty? || filter_data.include?(id)
|
||||
queries.each do |query|
|
||||
@formats = @formats.push(query)
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
# Just return datasource name
|
||||
# @return string
|
||||
def to_s
|
||||
DATASOURCE_NAME
|
||||
end
|
||||
|
||||
# If this datasource accepts a data import instance
|
||||
# @return Boolean
|
||||
def persists_state_via_data_import?
|
||||
false
|
||||
end
|
||||
|
||||
# Stores the data import item instance to use/manipulate it
|
||||
# @param value DataImport
|
||||
def data_import_item=(value)
|
||||
nil
|
||||
end
|
||||
|
||||
# Checks if token is still valid or has been revoked
|
||||
# @return bool
|
||||
# @throws AuthError
|
||||
def token_valid?
|
||||
# Any call would do, we just want to see if communicates or refuses the token
|
||||
@client.get_current_account
|
||||
true
|
||||
rescue DropboxApi::Errors::HttpError => ex
|
||||
CartoDB::Logger.debug(message: 'Invalid Dropbox token', exception: ex, user: @user)
|
||||
false
|
||||
end
|
||||
|
||||
# Revokes current set token
|
||||
def revoke_token
|
||||
@client.revoke
|
||||
true
|
||||
rescue DropboxApi::Errors::HttpError => ex
|
||||
CartoDB::Logger.debug(message: 'Error revoking Dropbox token', exception: ex, user: @user)
|
||||
true
|
||||
rescue => ex
|
||||
raise AuthError.new("revoke_token: #{ex.message}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
SEARCH_BATCH_SIZE = 1000
|
||||
|
||||
# Handles
|
||||
# @param original_exception mixed
|
||||
# @param message string
|
||||
# @throws TokenExpiredOrInvalidError
|
||||
# @throws AuthError
|
||||
# @throws mixed
|
||||
def handle_error(original_exception, message)
|
||||
if original_exception.is_a? DropboxApi::Errors::NotFoundError
|
||||
raise NotFoundDownloadError.new(message, DATASOURCE_NAME)
|
||||
elsif original_exception.is_a? DropboxApi::Errors::BasicError
|
||||
error_code = original_exception.http_response.code.to_i
|
||||
if error_code == 401 || error_code == 403
|
||||
raise TokenExpiredOrInvalidError.new(message, DATASOURCE_NAME)
|
||||
else
|
||||
raise AuthError.new(message)
|
||||
end
|
||||
elsif original_exception.is_a? ArgumentError
|
||||
raise DataDownloadError.new(message, DATASOURCE_NAME)
|
||||
else
|
||||
raise original_exception
|
||||
end
|
||||
end
|
||||
|
||||
# Formats all data to comply with our desired format
|
||||
# @param item_data Hash : Single item returned from Dropbox API
|
||||
# @return { :id, :title, :url, :service, :size }
|
||||
def format_item_data(resource)
|
||||
path = resource.path_display
|
||||
filename = path.split('/').last
|
||||
|
||||
{
|
||||
id: path,
|
||||
title: filename,
|
||||
filename: filename,
|
||||
service: DATASOURCE_NAME,
|
||||
checksum: checksum_of(resource.rev),
|
||||
size: resource.size
|
||||
}
|
||||
end
|
||||
|
||||
def authenticator
|
||||
@authenticator ||= DropboxApi::Authenticator.new(@app_key, @app_secret)
|
||||
end
|
||||
|
||||
def state
|
||||
service_name = service_name_for_user(DATASOURCE_NAME, @user)
|
||||
CALLBACK_STATE_DATA_PLACEHOLDER.sub('user', @user.username).sub('service', service_name)
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,353 @@
|
||||
require 'signet/oauth_2/client'
|
||||
require 'google/apis/drive_v2'
|
||||
require_relative '../../../../../lib/carto/http/client'
|
||||
|
||||
module CartoDB
|
||||
module Datasources
|
||||
module Url
|
||||
class GDrive < BaseOAuth
|
||||
|
||||
# Required for all providers
|
||||
DATASOURCE_NAME = 'gdrive'
|
||||
|
||||
OAUTH_SCOPES = ['https://www.googleapis.com/auth/drive'].freeze
|
||||
# For when using authorization code instead of callback with token
|
||||
REDIRECT_URI = 'urn:ietf:wg:oauth:2.0:oob'
|
||||
FIELDS_TO_RETRIEVE = 'items(downloadUrl,exportLinks,id,modifiedDate,title,fileExtension,fileSize)'
|
||||
|
||||
# Specific of this provider
|
||||
FORMATS_TO_MIME_TYPES = {
|
||||
FORMAT_CSV => %w(text/csv),
|
||||
FORMAT_EXCEL => %w(application/vnd.ms-excel application/vnd.google-apps.spreadsheet application/vnd.openxmlformats-officedocument.spreadsheetml.sheet),
|
||||
# FORMAT_GPX => %w(text/xml), # Disabled because text/xml list any XML file
|
||||
FORMAT_KML => %w(application/vnd.google-earth.kml+xml),
|
||||
FORMAT_COMPRESSED => %w(application/zip application/x-zip-compressed), # application/x-compressed-tar application/x-gzip application/x-bzip application/x-tar )
|
||||
}
|
||||
|
||||
# Constructor (hidden)
|
||||
# @param config
|
||||
# [
|
||||
# 'application_name'
|
||||
# 'client_id'
|
||||
# 'client_secret'
|
||||
# ]
|
||||
# @param user ::User
|
||||
# @throws UninitializedError
|
||||
# @throws MissingConfigurationError
|
||||
def initialize(config, user)
|
||||
super(config, user, %w{ application_name client_id client_secret callback_url }, DATASOURCE_NAME)
|
||||
|
||||
raise UninitializedError.new('missing user instance', DATASOURCE_NAME) if user.nil?
|
||||
|
||||
self.filter=[]
|
||||
@refresh_token = nil
|
||||
|
||||
@user = user
|
||||
@callback_url = config.fetch('callback_url')
|
||||
@client = Signet::OAuth2::Client.new(
|
||||
authorization_uri: 'https://accounts.google.com/o/oauth2/auth',
|
||||
token_credential_uri: 'https://oauth2.googleapis.com/token',
|
||||
client_id: config.fetch('client_id'),
|
||||
client_secret: config.fetch('client_secret'),
|
||||
scope: OAUTH_SCOPES,
|
||||
redirect_uri: @callback_url,
|
||||
access_type: :offline
|
||||
)
|
||||
@drive = Google::Apis::DriveV2::DriveService.new
|
||||
@drive.authorization = @client
|
||||
end
|
||||
|
||||
# Factory method
|
||||
# @param config {}
|
||||
# @param user ::User
|
||||
# @return CartoDB::Datasources::Url::GDrive
|
||||
def self.get_new(config, user)
|
||||
new(config, user)
|
||||
end
|
||||
|
||||
# If will provide a url to download the resource, or requires calling get_resource()
|
||||
# @return bool
|
||||
def providers_download_url?
|
||||
false
|
||||
end
|
||||
|
||||
# Return the url to be displayed or sent the user to authenticate and get authorization code
|
||||
# @param use_callback_flow : bool
|
||||
# @return string | nil
|
||||
def get_auth_url(use_callback_flow = true)
|
||||
if use_callback_flow
|
||||
service_name = service_name_for_user(DATASOURCE_NAME, @user)
|
||||
@client.state = CALLBACK_STATE_DATA_PLACEHOLDER.sub('user', @user.username)
|
||||
.sub('service', service_name)
|
||||
else
|
||||
@client.redirect_uri = REDIRECT_URI
|
||||
end
|
||||
@client.authorization_uri.to_s
|
||||
end
|
||||
|
||||
# Validate authorization code and store token
|
||||
# @param auth_code : string
|
||||
# @param use_callback_flow : bool
|
||||
# @return string : Access token
|
||||
# @throws AuthError
|
||||
def validate_auth_code(auth_code, use_callback_flow = true)
|
||||
unless use_callback_flow
|
||||
@client.redirect_uri = REDIRECT_URI
|
||||
end
|
||||
@client.code = auth_code
|
||||
@client.fetch_access_token!
|
||||
if @client.refresh_token.nil?
|
||||
raise AuthError.new(
|
||||
"Error validating auth token. Is this Google account linked to another CARTO account?",
|
||||
DATASOURCE_NAME
|
||||
)
|
||||
end
|
||||
@refresh_token = @client.refresh_token
|
||||
rescue Google::Apis::AuthorizationError, Signet::AuthorizationError => ex
|
||||
raise AuthError.new("validating auth code: #{ex.message}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
# Validates the authorization callback
|
||||
# @param params : mixed
|
||||
def validate_callback(params)
|
||||
if params[:error].present?
|
||||
raise AuthError.new("validate_callback: #{params[:error]}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
if params[:code]
|
||||
validate_auth_code(params[:code])
|
||||
else
|
||||
raise AuthError.new('validate_callback: Missing authorization code', DATASOURCE_NAME)
|
||||
end
|
||||
end
|
||||
|
||||
# Store token
|
||||
# Triggers generation of a valid access token for the lifetime of this instance
|
||||
# @param token string
|
||||
# @throws AuthError
|
||||
def token=(token)
|
||||
@refresh_token = token
|
||||
@client.update_token!(refresh_token: @refresh_token)
|
||||
@client.fetch_access_token!
|
||||
rescue Google::Apis::AuthorizationError, Signet::AuthorizationError => ex
|
||||
raise TokenExpiredOrInvalidError.new("Invalid token: #{ex.message}", DATASOURCE_NAME)
|
||||
rescue Google::Apis::ClientError, \
|
||||
Google::Apis::ServerError, Google::Apis::BatchError, Google::Apis::TransmissionError => ex
|
||||
raise AuthError.new("setting token: #{ex.message}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
# Retrieve token
|
||||
# @return string | nil
|
||||
def token
|
||||
@refresh_token
|
||||
end
|
||||
|
||||
# Perform the listing and return results
|
||||
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
|
||||
# @return [ { :id, :title, :url, :service } ]
|
||||
# @throws TokenExpiredOrInvalidError
|
||||
# @throws DataDownloadError
|
||||
def get_resources_list(filter=[])
|
||||
all_results = []
|
||||
self.filter = filter
|
||||
|
||||
@drive.batch do |d|
|
||||
@formats.each do |mime_type|
|
||||
d.list_files(q: "mime_type = '#{mime_type}'", fields: FIELDS_TO_RETRIEVE) do |res, err|
|
||||
if err
|
||||
case err.status_code
|
||||
when 200
|
||||
break
|
||||
when 403
|
||||
raise GDriveNoExternalAppsAllowedError.new(result.data['error']['message'], DATASOURCE_NAME)
|
||||
else
|
||||
error_msg = "get_resources_list() #{result.data['error']['message']} (#{result.status})"
|
||||
raise DataDownloadError.new(error_msg, DATASOURCE_NAME)
|
||||
end
|
||||
elsif res.items.present?
|
||||
res.items.each do |item|
|
||||
all_results.push(format_item_data(item))
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
all_results.compact
|
||||
rescue Google::Apis::AuthorizationError, Signet::AuthorizationError => ex
|
||||
raise TokenExpiredOrInvalidError.new("Invalid token: #{ex.message}", DATASOURCE_NAME)
|
||||
rescue Google::Apis::BatchError, Google::Apis::TransmissionError, Google::Apis::ClientError, \
|
||||
Google::Apis::ServerError => ex
|
||||
raise DataDownloadError.new("getting resources: #{ex.message}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
# Retrieves a resource and returns its contents
|
||||
# @param id string
|
||||
# @return mixed
|
||||
# @throws TokenExpiredOrInvalidError
|
||||
# @throws DataDownloadError
|
||||
def get_resource(id)
|
||||
@drive.get_file(id) do |file, err|
|
||||
if err
|
||||
error_msg = "(#{err.status_code}) retrieving file #{id}: #{err}"
|
||||
raise DataDownloadError.new(error_msg, DATASOURCE_NAME)
|
||||
end
|
||||
if file.export_links.present?
|
||||
@drive.export_file(file.id, 'text/csv', download_dest: StringIO.new) do |content, export_err|
|
||||
raise export_err if export_err
|
||||
return content
|
||||
end
|
||||
else
|
||||
@drive.get_file(file.id, download_dest: StringIO.new) do |content, download_err|
|
||||
raise download_err if download_err
|
||||
return content
|
||||
end
|
||||
end
|
||||
end
|
||||
rescue Google::Apis::AuthorizationError, Signet::AuthorizationError => ex
|
||||
raise TokenExpiredOrInvalidError.new("Invalid token: #{ex.message}", DATASOURCE_NAME)
|
||||
rescue Google::Apis::BatchError, Google::Apis::TransmissionError, Google::Apis::ClientError, \
|
||||
Google::Apis::ServerError => ex
|
||||
raise DataDownloadError.new("downloading file #{id}: #{ex.message}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
# @param id string
|
||||
# @return Hash
|
||||
# @throws TokenExpiredOrInvalidError
|
||||
# @throws DataDownloadError
|
||||
# @throws NotFoundDownloadError
|
||||
def get_resource_metadata(id)
|
||||
@drive.get_file(id) do |file, err|
|
||||
if err
|
||||
case err.status_code
|
||||
when 404
|
||||
error_msg = "(#{err.status_code}) retrieving file #{id} metadata: #{err}, should stop syncing"
|
||||
raise NotFoundDownloadError.new(error_msg, DATASOURCE_NAME)
|
||||
else
|
||||
error_msg = "(#{err.status_code}) retrieving file #{id} metadata: #{err}"
|
||||
raise DataDownloadError.new(error_msg, DATASOURCE_NAME)
|
||||
end
|
||||
else
|
||||
item_data = format_item_data(file)
|
||||
return item_data.to_hash
|
||||
end
|
||||
end
|
||||
rescue Google::Apis::AuthorizationError, Signet::AuthorizationError
|
||||
raise TokenExpiredOrInvalidError.new('Invalid token', DATASOURCE_NAME)
|
||||
rescue Google::Apis::BatchError, Google::Apis::TransmissionError, Google::Apis::ClientError, \
|
||||
Google::Apis::ServerError
|
||||
raise DataDownloadError.new("get_resource_metadata() #{id}", DATASOURCE_NAME)
|
||||
rescue StandardError => e
|
||||
CartoDB.notify_exception(e, id: id, user: @user)
|
||||
raise e
|
||||
end
|
||||
|
||||
# Retrieves current filters
|
||||
# @return {}
|
||||
def filter
|
||||
@formats
|
||||
end
|
||||
|
||||
# Sets current filters
|
||||
# @param filter_data {}
|
||||
def filter=(filter_data=[])
|
||||
@formats = []
|
||||
FORMATS_TO_MIME_TYPES.each do |id, mime_types|
|
||||
if filter_data.empty? || filter_data.include?(id)
|
||||
mime_types.each do |mime_type|
|
||||
@formats = @formats.push(mime_type)
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
# Just return datasource name
|
||||
# @return string
|
||||
def to_s
|
||||
DATASOURCE_NAME
|
||||
end
|
||||
|
||||
# If this datasource accepts a data import instance
|
||||
# @return Boolean
|
||||
def persists_state_via_data_import?
|
||||
false
|
||||
end
|
||||
|
||||
# Stores the data import item instance to use/manipulate it
|
||||
# @param value DataImport
|
||||
def data_import_item=(value)
|
||||
nil
|
||||
end
|
||||
|
||||
# Checks if token is still valid or has been revoked
|
||||
# @return bool
|
||||
# @throws AuthError
|
||||
def token_valid?
|
||||
# Any call would do, we just want to see if communicates or refuses the token
|
||||
result = @drive.get_about
|
||||
!result.nil?
|
||||
rescue Google::Apis::AuthorizationError, Signet::AuthorizationError
|
||||
false
|
||||
rescue Google::Apis::BatchError, Google::Apis::TransmissionError, Google::Apis::ClientError, \
|
||||
Google::Apis::ServerError => ex
|
||||
raise AuthError.new("token_valid?() #{id}: #{ex.message}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
# Revokes current set token
|
||||
def revoke_token
|
||||
http_client = Carto::Http::Client.get('gdrive',
|
||||
connecttimeout: 60,
|
||||
timeout: 600)
|
||||
response = http_client.get("https://accounts.google.com/o/oauth2/revoke?token=#{token}")
|
||||
if response.code == 200
|
||||
true
|
||||
end
|
||||
rescue StandardError => ex
|
||||
raise AuthError.new("revoke_token: #{ex.message}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
# Formats all data to comply with our desired format
|
||||
# @param item_data Hash : Single item returned from GDrive API
|
||||
# @return { :id, :title, :url, :service, :checksum, :size }
|
||||
def format_item_data(item_data)
|
||||
data =
|
||||
{
|
||||
id: item_data.id,
|
||||
title: item_data.title,
|
||||
service: DATASOURCE_NAME,
|
||||
checksum: checksum_of(item_data.modified_date.to_s)
|
||||
|
||||
}
|
||||
if item_data.export_links.present?
|
||||
# Native spreadsheets have no format nor direct download links
|
||||
data[:url] = item_data.export_links.first.last
|
||||
data[:url] = data[:url][0..data[:url].rindex('=')] + 'csv'
|
||||
data[:filename] = clean_filename(item_data.title) + '.csv'
|
||||
data[:size] = NO_CONTENT_SIZE_PROVIDED
|
||||
elsif item_data.download_url.present?
|
||||
data[:url] = item_data.download_url
|
||||
# For Drive files, title == filename + extension
|
||||
data[:filename] = item_data.title
|
||||
data[:size] = item_data.file_size.to_i
|
||||
else
|
||||
# Downloads from files shared by other people can be disabled, ignore them
|
||||
return nil
|
||||
end
|
||||
data
|
||||
end
|
||||
|
||||
def clean_filename(name)
|
||||
clean_name = ''
|
||||
name.gsub(' ','_').scan(/([a-zA-Z0-9_]+)/).flatten.map { |match|
|
||||
clean_name << match
|
||||
}
|
||||
clean_name = name if clean_name.size == 0
|
||||
|
||||
clean_name
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,287 @@
|
||||
require "instagram"
|
||||
|
||||
module CartoDB
|
||||
module Datasources
|
||||
module Url
|
||||
class InstagramOAuth < BaseOAuth
|
||||
|
||||
# Required for all datasources
|
||||
DATASOURCE_NAME = 'instagram'
|
||||
|
||||
FORMAT_ALL_MEDIA = 'all_media'
|
||||
|
||||
# Constructor
|
||||
# @param config Array
|
||||
# [
|
||||
# 'app_key'
|
||||
# 'app_secret'
|
||||
# 'callback_url'
|
||||
# ]
|
||||
# @param user ::User
|
||||
# @throws UninitializedError
|
||||
# @throws MissingConfigurationError
|
||||
def initialize(config, user)
|
||||
super(config, user, %w{ app_key app_secret callback_url }, DATASOURCE_NAME)
|
||||
|
||||
@user = user
|
||||
@app_key = config.fetch('app_key')
|
||||
@app_secret = config.fetch('app_secret')
|
||||
|
||||
raise ServiceDisabledError.new(DATASOURCE_NAME, @user.username) unless @user.has_feature_flag?('instagram_import')
|
||||
|
||||
service_name = service_name_for_user(DATASOURCE_NAME, @user)
|
||||
placeholder = CALLBACK_STATE_DATA_PLACEHOLDER.sub('user', @user.username).sub('service', service_name)
|
||||
@callback_url = "#{config.fetch('callback_url')}?state=#{placeholder}"
|
||||
|
||||
self.filter = []
|
||||
@access_token = nil
|
||||
@auth_flow = nil
|
||||
@client = nil
|
||||
end
|
||||
|
||||
# Factory method
|
||||
# @param config : {}
|
||||
# @param user : ::User
|
||||
# @return CartoDB::Datasources::Url::InstagramOAuth
|
||||
def self.get_new(config, user)
|
||||
return new(config, user)
|
||||
end
|
||||
|
||||
# If will provide a url to download the resource, or requires calling get_resource()
|
||||
# @return bool
|
||||
def providers_download_url?
|
||||
false
|
||||
end
|
||||
|
||||
# Return the url to be displayed or sent the user to to authenticate and get authorization code
|
||||
# @param use_callback_flow : bool
|
||||
# @throws AuthError
|
||||
def get_auth_url(use_callback_flow=true)
|
||||
# TODO: Add CSRF here (http://instagram.com/developer/authentication/)
|
||||
Instagram.authorize_url({
|
||||
client_id: @app_key,
|
||||
response_type: 'code',
|
||||
redirect_uri: @callback_url
|
||||
})
|
||||
rescue => ex
|
||||
raise AuthError.new("get_auth_url(#{use_callback_flow}): #{ex.message}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
# Validates the authorization callback
|
||||
# @param params : mixed
|
||||
def validate_callback(params)
|
||||
response = Instagram.get_access_token(params[:code], {
|
||||
client_id: @app_key,
|
||||
client_secret: @app_secret,
|
||||
redirect_uri: @callback_url
|
||||
})
|
||||
@access_token = response.access_token
|
||||
@client = Instagram.client(access_token: @access_token)
|
||||
@access_token
|
||||
rescue => ex
|
||||
raise AuthError.new("validate_callback(#{params.inspect}): #{ex.message}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
# Set the token
|
||||
# @param token string
|
||||
# @throws TokenExpiredOrInvalidError
|
||||
# @throws AuthError
|
||||
def token=(token)
|
||||
@access_token = token
|
||||
@client = Instagram.client(access_token: @access_token)
|
||||
rescue => ex
|
||||
handle_error(ex, "token= : #{ex.message}")
|
||||
end
|
||||
|
||||
# Retrieve set token
|
||||
# @return string | nil
|
||||
def token
|
||||
@access_token
|
||||
end
|
||||
|
||||
# Perform the listing and return results
|
||||
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
|
||||
# @return [ { :id, :title, :url, :service } ]
|
||||
# @throws TokenExpiredOrInvalidError
|
||||
# @throws AuthError
|
||||
# @throws DataDownloadError
|
||||
def get_resources_list(filter=[])
|
||||
[
|
||||
{
|
||||
id: FORMAT_ALL_MEDIA,
|
||||
title: 'All your photos and videos',
|
||||
url: 'All your photos and videos',
|
||||
service: DATASOURCE_NAME,
|
||||
checksum: '',
|
||||
size: NO_CONTENT_SIZE_PROVIDED
|
||||
}
|
||||
]
|
||||
end
|
||||
|
||||
# Retrieves a resource and returns its contents
|
||||
# @param id string
|
||||
# @return mixed
|
||||
# @throws TokenExpiredOrInvalidError
|
||||
# @throws AuthError
|
||||
# @throws DataDownloadError
|
||||
def get_resource(id)
|
||||
|
||||
contents = [
|
||||
field_to_csv('thumbnail'),
|
||||
field_to_csv('image'),
|
||||
field_to_csv('link'),
|
||||
field_to_csv('type'),
|
||||
field_to_csv('lat'),
|
||||
field_to_csv('lon'),
|
||||
field_to_csv('location_id'),
|
||||
field_to_csv('location_name'),
|
||||
field_to_csv('caption'),
|
||||
field_to_csv('comments_count'),
|
||||
field_to_csv('likes_count'),
|
||||
field_to_csv('tags'),
|
||||
field_to_csv('created_time')
|
||||
].join(',') << "\n"
|
||||
|
||||
max_id = nil
|
||||
|
||||
begin
|
||||
batch_contents, max_id = get_resource_page(id, max_id)
|
||||
contents << batch_contents
|
||||
end while !max_id.nil?
|
||||
|
||||
contents
|
||||
end
|
||||
|
||||
# @param id string
|
||||
# @return Hash
|
||||
# @throws TokenExpiredOrInvalidError
|
||||
# @throws AuthError
|
||||
# @throws DataDownloadError
|
||||
def get_resource_metadata(id)
|
||||
{
|
||||
id: FORMAT_ALL_MEDIA,
|
||||
filename: "#{DATASOURCE_NAME}_#{@client.user.username}.csv",
|
||||
size: NO_CONTENT_SIZE_PROVIDED
|
||||
}
|
||||
rescue => ex
|
||||
handle_error(ex, "get_resource_metadata() #{id}: #{ex.message}")
|
||||
end
|
||||
|
||||
# Retrieves current filters
|
||||
# @return {}
|
||||
def filter
|
||||
{}
|
||||
end
|
||||
|
||||
# Sets current filters
|
||||
# @param filter_data {}
|
||||
def filter=(filter_data=[])
|
||||
nil
|
||||
end
|
||||
|
||||
# Just return datasource name
|
||||
# @return string
|
||||
def to_s
|
||||
DATASOURCE_NAME
|
||||
end
|
||||
|
||||
# If this datasource accepts a data import instance
|
||||
# @return Boolean
|
||||
def persists_state_via_data_import?
|
||||
false
|
||||
end
|
||||
|
||||
# Stores the data import item instance to use/manipulate it
|
||||
# @param value DataImport
|
||||
def data_import_item=(value)
|
||||
nil
|
||||
end
|
||||
|
||||
# Checks if token is still valid or has been revoked
|
||||
# @return bool
|
||||
# @throws AuthError
|
||||
def token_valid?
|
||||
# checking if metadata is returned, if so token
|
||||
# is valid, if not it is invalid
|
||||
response = get_resource_metadata(DATASOURCE_NAME)
|
||||
if response[:id]
|
||||
true
|
||||
end
|
||||
rescue => ex
|
||||
false
|
||||
end
|
||||
|
||||
# Revokes current set token
|
||||
def revoke_token
|
||||
# TODO: See how to check this
|
||||
true
|
||||
rescue => ex
|
||||
raise AuthError.new("revoke_token: #{ex.message}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def field_to_csv(field)
|
||||
'"' + field.to_s.gsub('"', '""').gsub("\\n", ' ').gsub("\x0D", ' ').gsub("\x0A", ' ').gsub("\0", '')
|
||||
.gsub("\\", ' ') + '"'
|
||||
end
|
||||
|
||||
# @param resource_id String
|
||||
# @para max_id Integer|nil Max media id retrieved (used to paginate)
|
||||
def get_resource_page(resource_id, max_id=nil)
|
||||
contents = ''
|
||||
|
||||
data = { count: 30 }
|
||||
data[:max_id] = max_id unless max_id.nil?
|
||||
|
||||
items = @client.user_recent_media(data)
|
||||
new_max_id = items.pagination.next_max_id
|
||||
|
||||
items.each do |item|
|
||||
if item.location.nil?
|
||||
lat = lon = location_id = location_name = nil
|
||||
else
|
||||
lat = item.location.latitude
|
||||
lon = item.location.longitude
|
||||
location_id = item.location.id
|
||||
location_name = item.location.name
|
||||
end
|
||||
caption = item.caption.nil? ? '' : item.caption.text
|
||||
|
||||
contents << [
|
||||
field_to_csv(item.images.thumbnail.url),
|
||||
field_to_csv(item.images.standard_resolution.url),
|
||||
field_to_csv(item.link),
|
||||
field_to_csv(item.type),
|
||||
field_to_csv(lat),
|
||||
field_to_csv(lon),
|
||||
field_to_csv(location_id),
|
||||
field_to_csv(location_name),
|
||||
field_to_csv(caption),
|
||||
field_to_csv(item.comments['count']),
|
||||
field_to_csv(item.likes['count']),
|
||||
field_to_csv(item.tags.join(',')),
|
||||
field_to_csv(item.created_time)
|
||||
].join(',') << "\n"
|
||||
end
|
||||
|
||||
[ contents, new_max_id ]
|
||||
rescue => ex
|
||||
handle_error(ex, "get_resource() #{resource_id}: #{ex.message}")
|
||||
end
|
||||
|
||||
# Handles
|
||||
# @param original_exception mixed
|
||||
# @param message string
|
||||
# @throws TokenExpiredOrInvalidError
|
||||
# @throws AuthError
|
||||
# @throws mixed
|
||||
def handle_error(original_exception, message)
|
||||
# TODO: Implement
|
||||
raise original_exception
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,397 @@
|
||||
require 'json'
|
||||
require 'gibbon'
|
||||
require 'addressable/uri'
|
||||
require_relative '../base_oauth'
|
||||
require_relative '../../../../../lib/carto/http/client'
|
||||
|
||||
module CartoDB
|
||||
module Datasources
|
||||
module Url
|
||||
# Note:
|
||||
# - MailChimp access tokens don't expire, no need to handle that logic
|
||||
class MailChimp < BaseOAuth
|
||||
|
||||
# Required for all datasources
|
||||
DATASOURCE_NAME = 'mailchimp'
|
||||
|
||||
AUTHORIZE_URI = 'https://login.mailchimp.com/oauth2/authorize?response_type=code&client_id=%s&redirect_uri=%s'
|
||||
ACCESS_TOKEN_URI = 'https://login.mailchimp.com/oauth2/token'
|
||||
MAILCHIMP_METADATA_URI = 'https://login.mailchimp.com/oauth2/metadata'
|
||||
|
||||
API_TIMEOUT_SECS = 60
|
||||
|
||||
# Constructor
|
||||
# @param config Array
|
||||
# [
|
||||
# 'api_key'
|
||||
# 'timeout_minutes'
|
||||
# ]
|
||||
# @param user ::User
|
||||
# @throws UninitializedError
|
||||
# @throws MissingConfigurationError
|
||||
def initialize(config, user)
|
||||
super(config, user, %w{ app_key app_secret callback_url }, DATASOURCE_NAME)
|
||||
|
||||
@user = user
|
||||
@app_key = config.fetch('app_key')
|
||||
@app_secret = config.fetch('app_secret')
|
||||
|
||||
@http_timeout = config.fetch(:http_timeout, 600)
|
||||
@http_connect_timeout = config.fetch(:http_connect_timeout, 60)
|
||||
|
||||
service_name = service_name_for_user(DATASOURCE_NAME, @user)
|
||||
placeholder = CALLBACK_STATE_DATA_PLACEHOLDER.sub('user', @user.username).sub('service', service_name)
|
||||
@callback_url = "#{config.fetch('callback_url')}?state=#{placeholder}"
|
||||
|
||||
Gibbon::API.timeout = API_TIMEOUT_SECS
|
||||
Gibbon::API.throws_exceptions = true
|
||||
Gibbon::Export.timeout = API_TIMEOUT_SECS
|
||||
Gibbon::Export.throws_exceptions = false
|
||||
|
||||
@access_token = nil
|
||||
@api_client = nil
|
||||
end
|
||||
|
||||
# Factory method
|
||||
# @param config : {}
|
||||
# @param user : ::User
|
||||
# @return CartoDB::Datasources::Url::MailChimpLists
|
||||
def self.get_new(config, user)
|
||||
return new(config, user)
|
||||
end
|
||||
|
||||
# If will provide a url to download the resource, or requires calling get_resource()
|
||||
# @return bool
|
||||
def providers_download_url?
|
||||
false
|
||||
end
|
||||
|
||||
# Return the url to be displayed or sent the user to to authenticate and get authorization code
|
||||
# @param use_callback_flow : bool
|
||||
# @return string : URL to navigate to for the authorization flow
|
||||
# @throws ExternalServiceError
|
||||
def get_auth_url(use_callback_flow=true)
|
||||
if use_callback_flow
|
||||
AUTHORIZE_URI % [@app_key, Addressable::URI.encode(@callback_url)]
|
||||
else
|
||||
raise ExternalServiceError.new("This datasource doesn't allows non-callback flows", DATASOURCE_NAME)
|
||||
end
|
||||
end
|
||||
|
||||
# Validate authorization code and store token
|
||||
# @param auth_code : string
|
||||
# @return string : Access token
|
||||
# @throws ExternalServiceError
|
||||
def validate_auth_code(auth_code)
|
||||
raise ExternalServiceError.new("This datasource doesn't allows non-callback flows", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
# Validates the authorization callback
|
||||
# @param params : mixed
|
||||
# @throws AuthError
|
||||
# @throws DataDownloadTimeoutError
|
||||
def validate_callback(params)
|
||||
code = params.fetch('code')
|
||||
if code.nil? || code == ''
|
||||
raise "Empty callback code"
|
||||
end
|
||||
|
||||
token_call_params = {
|
||||
grant_type: 'authorization_code',
|
||||
client_id: @app_key,
|
||||
client_secret: @app_secret,
|
||||
code: code,
|
||||
redirect_uri: @callback_url
|
||||
}
|
||||
|
||||
token_response = http_client.post(ACCESS_TOKEN_URI, http_options(token_call_params, :post))
|
||||
|
||||
raise DataDownloadTimeoutError.new(DATASOURCE_NAME) if token_response.timed_out?
|
||||
|
||||
unless token_response.code == 200
|
||||
raise "Bad token response: #{token_response.body.inspect} (#{token_response.code})"
|
||||
end
|
||||
token_data = ::JSON.parse(token_response.body)
|
||||
|
||||
partial_access_token = token_data['access_token']
|
||||
|
||||
# Afterwards, must do another call to metadata endpoint to retrieve API details
|
||||
# @see https://apidocs.mailchimp.com/oauth2/
|
||||
metadata_response = http_client.get(MAILCHIMP_METADATA_URI,http_options({}, :get, {
|
||||
'Authorization' => "OAuth #{partial_access_token}"}))
|
||||
|
||||
raise DataDownloadTimeoutError.new(DATASOURCE_NAME) if metadata_response.timed_out?
|
||||
|
||||
unless metadata_response.code == 200
|
||||
raise "Bad metadata response: #{metadata_response.body.inspect} (#{metadata_response.code})"
|
||||
end
|
||||
metadata_data = ::JSON.parse(metadata_response.body)
|
||||
|
||||
# This specially formed token behaves as an API Key for client calls using API
|
||||
@access_token = "#{partial_access_token}-#{metadata_data['dc']}"
|
||||
rescue => ex
|
||||
raise AuthError.new("validate_callback(#{params.inspect}): #{ex.message}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
# Set the token
|
||||
# @param token string
|
||||
# @throws TokenExpiredOrInvalidError
|
||||
def token=(token)
|
||||
@access_token = token
|
||||
@api_client = Gibbon::API.new(@access_token)
|
||||
rescue Gibbon::MailChimpError => exception
|
||||
raise TokenExpiredOrInvalidError.new("token=() : #{exception.message} (API code: #{exception.code})",
|
||||
DATASOURCE_NAME)
|
||||
rescue => exception
|
||||
raise TokenExpiredOrInvalidError.new("token=() : #{exception.inspect}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
# Retrieve set token
|
||||
# @return string | nil
|
||||
def token
|
||||
@access_token
|
||||
end
|
||||
|
||||
# Perform the listing and return results
|
||||
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
|
||||
# @return [ { :id, :title, :url, :service } ]
|
||||
# @throws UninitializedError
|
||||
# @throws DataDownloadError
|
||||
def get_resources_list(filter=[])
|
||||
raise UninitializedError.new('No API client instantiated', DATASOURCE_NAME) unless @api_client.present?
|
||||
|
||||
all_results = []
|
||||
offset = 0
|
||||
limit = 100
|
||||
total = nil
|
||||
|
||||
begin
|
||||
response = @api_client.campaigns.list({
|
||||
start: offset,
|
||||
limit: limit
|
||||
})
|
||||
errors = response.fetch('errors', [])
|
||||
unless errors.empty?
|
||||
raise DataDownloadError.new("get_resources_list(): #{errors.inspect}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
total = response.fetch('total', 0).to_i if total.nil?
|
||||
|
||||
response_data = response.fetch('data', [])
|
||||
response_data.each do |item|
|
||||
# Skip items without tracking
|
||||
all_results.push(format_activity_item_data(item)) if item['tracking']['opens']
|
||||
end
|
||||
|
||||
offset += limit
|
||||
end while offset < total
|
||||
|
||||
all_results
|
||||
rescue Gibbon::MailChimpError => exception
|
||||
raise DataDownloadError.new("get_resources_list(): #{exception.message} (API code: #{exception.code}",
|
||||
DATASOURCE_NAME)
|
||||
rescue => exception
|
||||
raise DataDownloadError.new("get_resources_list(): #{exception.inspect}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
# Retrieves a resource and returns its contents
|
||||
# @param id string
|
||||
# @return mixed
|
||||
# @throws UninitializedError
|
||||
# @throws DataDownloadError
|
||||
def get_resource(id)
|
||||
raise UninitializedError.new('No API client instantiated', DATASOURCE_NAME) unless @api_client.present?
|
||||
|
||||
subscribers = {}
|
||||
contents = StringIO.new
|
||||
export_api = @api_client.get_exporter
|
||||
|
||||
# 1) Retrieve campaign details
|
||||
campaign = get_resource_metadata(id)
|
||||
campaign_details = export_api.list({id: campaign[:list_id]})
|
||||
campaign = nil
|
||||
|
||||
# 2) Retrieve subscriber activity
|
||||
# https://apidocs.mailchimp.com/export/1.0/campaignsubscriberactivity.func.php
|
||||
subscribers_activity = export_api.campaign_subscriber_activity({id: id})
|
||||
|
||||
subscribers_activity.each { |line|
|
||||
store_subscriber_if_opened(line, subscribers)
|
||||
}
|
||||
subscribers_activity = nil
|
||||
|
||||
# 3) Update campaign details with subscriber activity results
|
||||
# 4) anonymize data (inside list_json_to_csv)
|
||||
campaign_details.each_with_index { |line, index|
|
||||
contents.write list_json_to_csv(line, subscribers, index == 0)
|
||||
}
|
||||
|
||||
contents.string
|
||||
rescue Gibbon::MailChimpError => exception
|
||||
raise DataDownloadError.new("get_resource(): #{exception.message} (API code: #{exception.code}",
|
||||
DATASOURCE_NAME)
|
||||
rescue => exception
|
||||
raise DataDownloadError.new("get_resource(): #{exception.inspect}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
# @param id string
|
||||
# @return Hash
|
||||
# @throws UninitializedError
|
||||
# @throws DataDownloadError
|
||||
def get_resource_metadata(id)
|
||||
raise UninitializedError.new('No API client instantiated', DATASOURCE_NAME) unless @api_client.present?
|
||||
|
||||
item_data = {}
|
||||
|
||||
# No metadata call at API, so just retrieve same info but from specific campaign id
|
||||
# https://apidocs.mailchimp.com/api/2.0/campaigns/list.php
|
||||
response = @api_client.campaigns.list({ filters: { campaign_id: id } })
|
||||
|
||||
errors = response.fetch('errors', [])
|
||||
unless errors.empty?
|
||||
raise DataDownloadError.new("get_resources_list(): #{errors.inspect}", DATASOURCE_NAME)
|
||||
end
|
||||
response_data = response.fetch('data', [])
|
||||
|
||||
response_data.each do |item|
|
||||
if item.fetch('id') == id
|
||||
item_data = format_activity_item_data(item)
|
||||
end
|
||||
end
|
||||
|
||||
item_data
|
||||
rescue Gibbon::MailChimpError => exception
|
||||
raise DataDownloadError.new("get_resource_metadata(): #{exception.message} (API code: #{exception.code}",
|
||||
DATASOURCE_NAME)
|
||||
rescue => exception
|
||||
raise DataDownloadError.new("get_resource_metadata(): #{exception.inspect}", DATASOURCE_NAME)
|
||||
end
|
||||
|
||||
# Retrieves current filters
|
||||
# @return {}
|
||||
def filter
|
||||
[]
|
||||
end
|
||||
|
||||
# Sets current filters
|
||||
# @param filter_data {}
|
||||
def filter=(filter_data=[])
|
||||
end
|
||||
|
||||
# Just return datasource name
|
||||
# @return string
|
||||
def to_s
|
||||
DATASOURCE_NAME
|
||||
end
|
||||
|
||||
# If this datasource accepts a data import instance
|
||||
# @return Boolean
|
||||
def persists_state_via_data_import?
|
||||
false
|
||||
end
|
||||
|
||||
# Stores the data import item instance to use/manipulate it
|
||||
# @param value DataImport
|
||||
def data_import_item=(value)
|
||||
nil
|
||||
end
|
||||
|
||||
# Checks if token is still valid or has been revoked
|
||||
# @return bool
|
||||
# @throws AuthError
|
||||
def token_valid?
|
||||
raise UninitializedError.new('No API client instantiated', DATASOURCE_NAME) unless @api_client.present?
|
||||
|
||||
# Any call would do, we just want to see if communicates or refuses the token
|
||||
# This call is available to all roles
|
||||
response = @api_client.users.profile
|
||||
# 'errors' only appears in failure scenarios, while 'username' only if went ok
|
||||
response.fetch('errors', nil).nil? && !response.fetch('username', nil).nil?
|
||||
rescue => ex
|
||||
CartoDB.notify_exception(ex)
|
||||
false
|
||||
end
|
||||
|
||||
# Revokes current set token
|
||||
def revoke_token
|
||||
# not supported
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def http_client
|
||||
@http_client ||= Carto::Http::Client.get('mailchimp')
|
||||
end
|
||||
|
||||
def http_options(params={}, method=:get, extra_headers={})
|
||||
{
|
||||
method: method,
|
||||
params: method == :get ? params : {},
|
||||
body: method == :post ? params : {},
|
||||
followlocation: true,
|
||||
ssl_verifypeer: false,
|
||||
headers: {
|
||||
'Accept' => 'application/json'
|
||||
}.merge(extra_headers),
|
||||
ssl_verifyhost: 0,
|
||||
connecttimeout: @http_connect_timeout,
|
||||
timeout: @http_timeout
|
||||
}
|
||||
end
|
||||
|
||||
# Formats all data to comply with our desired format
|
||||
# @param item_data Hash : Single item returned from MailChimp API
|
||||
# @return { :id, :title, :url, :service, :size }
|
||||
def format_activity_item_data(item_data)
|
||||
filename = item_data.fetch('title').gsub(' ', '_')
|
||||
{
|
||||
id: item_data.fetch('id'),
|
||||
list_id: item_data.fetch('list_id'),
|
||||
title: "#{item_data.fetch('title')}",
|
||||
filename: "#{filename}.csv",
|
||||
service: DATASOURCE_NAME,
|
||||
checksum: '',
|
||||
member_count: item_data.fetch('emails_sent'),
|
||||
size: NO_CONTENT_SIZE_PROVIDED
|
||||
}
|
||||
end
|
||||
|
||||
def store_subscriber_if_opened(input_fields='[]', subscribers)
|
||||
contents = ::JSON.parse(input_fields)
|
||||
contents.each { |subject, actions|
|
||||
unless actions.length == 0
|
||||
actions.each { |action|
|
||||
if action["action"] == "open"
|
||||
subscribers[subject] = true
|
||||
end
|
||||
opened_action = true
|
||||
}
|
||||
end
|
||||
}
|
||||
end
|
||||
|
||||
# @param contents String containing a JSON array of fields (data of campaign user/target)
|
||||
# @param subscribers Hash { subject => opened_email }
|
||||
# @param header_row Boolean
|
||||
# @return String Containing a CSV ready to dump to a file
|
||||
def list_json_to_csv(contents='[]', subscribers={}, header_row=false)
|
||||
# shorcut: Remove newlines and Anonymize email addresses before parsing to speed up
|
||||
contents = ::JSON.parse(contents.gsub("\n", ' ').gsub(/(\w|\.|\-)+@/, ""))
|
||||
|
||||
opened_mail = !subscribers[contents[0]].nil?
|
||||
|
||||
cleaned_contents = []
|
||||
#Once parsed, each row contains data like account code, company name, email, first name...
|
||||
contents.each_with_index { |field, index|
|
||||
# Remove double quotes to avoid CSV errors
|
||||
cleaned_contents[index] = "\"#{field.to_s.gsub('"', '""')}\""
|
||||
}
|
||||
cleaned_contents.push("\"#{header_row ? 'Opened' : opened_mail.to_s}\"")
|
||||
data = cleaned_contents.join(',')
|
||||
data << "\n"
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,181 @@
|
||||
require_relative '../../../../../lib/carto/http/client'
|
||||
|
||||
module CartoDB
|
||||
module Datasources
|
||||
module Url
|
||||
class PublicUrl < Base
|
||||
|
||||
# Required for all datasources
|
||||
DATASOURCE_NAME = 'public_url'
|
||||
|
||||
URL_REGEXP = %r{://}
|
||||
|
||||
# Constructor (hidden)
|
||||
# @param config
|
||||
# [ ]
|
||||
def initialize(config)
|
||||
super
|
||||
|
||||
@http_timeout = config.fetch(:http_timeout, 3200)
|
||||
@http_connect_timeout = config.fetch(:http_connect_timeout, 60)
|
||||
@service_name = DATASOURCE_NAME
|
||||
@headers = nil
|
||||
@response = nil
|
||||
end
|
||||
|
||||
# Factory method
|
||||
# @param config {}
|
||||
# @return CartoDB::Datasources::Url::PublicUrl
|
||||
def self.get_new(config={})
|
||||
return new(config)
|
||||
end
|
||||
|
||||
# If will provide a url to download the resource, or requires calling get_resource()
|
||||
# @return bool
|
||||
def providers_download_url?
|
||||
true
|
||||
end
|
||||
|
||||
def get_http_response_code
|
||||
@response.code if !@response.nil? && !@response.code.nil?
|
||||
end
|
||||
|
||||
# Perform the listing and return results
|
||||
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
|
||||
# @return [ { :id, :title, :url, :service } ]
|
||||
def get_resources_list(filter=[])
|
||||
nil
|
||||
end
|
||||
|
||||
# Retrieves a resource and returns its contents
|
||||
# @param id string
|
||||
# @return mixed
|
||||
# @throws DataDownloadTimeoutError
|
||||
# @throws DataDownloadError
|
||||
def get_resource(id)
|
||||
response = http_client.get(id, http_options)
|
||||
while response.headers['location']
|
||||
response = http_client.get(id, http_options)
|
||||
end
|
||||
|
||||
raise DataDownloadTimeoutError.new(DATASOURCE_NAME) if response.timed_out?
|
||||
|
||||
raise DataDownloadError.new("get_resource() #{id}", DATASOURCE_NAME) unless response.code.to_s =~ /\A[23]\d+/
|
||||
|
||||
# To be used in when try to retrieve the http response code
|
||||
@response = response
|
||||
|
||||
response.response_body
|
||||
end
|
||||
|
||||
# @param id string
|
||||
# @return Hash
|
||||
def get_resource_metadata(id)
|
||||
fetch_headers(id)
|
||||
{
|
||||
id: id,
|
||||
title: id,
|
||||
url: id,
|
||||
service: DATASOURCE_NAME,
|
||||
checksum: checksum_of(id, etag_header, last_modified_header),
|
||||
size: content_length_header
|
||||
# No need to use :filename nor file
|
||||
}
|
||||
end
|
||||
|
||||
# Fetches the headers for a given url
|
||||
# @throws DataDownloadError
|
||||
def fetch_headers(url)
|
||||
if url =~ URL_REGEXP
|
||||
response = http_client.head(url, http_options)
|
||||
|
||||
raise DataDownloadTimeoutError.new(DATASOURCE_NAME) if response.timed_out?
|
||||
|
||||
# For example S3 only allows one verb per signed url (we use GET) so won't allow HEAD, but it's ok
|
||||
@headers = (response.code.to_s =~ /\A[23]\d+/) ? response.headers : {}
|
||||
else
|
||||
@headers = {}
|
||||
end
|
||||
end
|
||||
|
||||
# Get the etag header if present
|
||||
# @return string
|
||||
# @throws UninitializedError
|
||||
def etag_header
|
||||
raise UninitializedError.new('headers not fetched', DATASOURCE_NAME) if @headers.nil?
|
||||
etag = @headers.fetch('ETag', nil)
|
||||
etag ||= @headers.fetch('Etag', nil)
|
||||
etag ||= @headers.fetch('etag', '')
|
||||
etag = etag.delete('"').delete("'") unless etag.empty?
|
||||
etag
|
||||
end
|
||||
|
||||
# Get the last modified header if present
|
||||
# @return string
|
||||
# @throws UninitializedError
|
||||
def last_modified_header
|
||||
raise UninitializedError.new('headers not fetched', DATASOURCE_NAME) if @headers.nil?
|
||||
last_modified = @headers.fetch('Last-Modified', nil)
|
||||
last_modified ||= @headers.fetch('Last-modified', nil)
|
||||
last_modified ||= @headers.fetch('last-modified', '')
|
||||
last_modified = last_modified.delete('"').delete("'") unless last_modified.empty?
|
||||
last_modified
|
||||
end
|
||||
|
||||
# Just return datasource name
|
||||
# @return string
|
||||
def to_s
|
||||
DATASOURCE_NAME
|
||||
end
|
||||
|
||||
# If this datasource accepts a data import instance
|
||||
# @return Boolean
|
||||
def persists_state_via_data_import?
|
||||
false
|
||||
end
|
||||
|
||||
# Stores the data import item instance to use/manipulate it
|
||||
# @param value DataImport
|
||||
def data_import_item=(value)
|
||||
nil
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def http_client
|
||||
@http_client ||= Carto::Http::Client.get('public_url')
|
||||
end
|
||||
|
||||
# Get the file size if present
|
||||
# @return Integer
|
||||
# @throws UninitializedError
|
||||
def content_length_header
|
||||
raise UninitializedError.new('headers not fetched', DATASOURCE_NAME) if @headers.nil?
|
||||
content_length = @headers.fetch('Content-Length', nil)
|
||||
content_length ||= @headers.fetch('Content-length', nil)
|
||||
content_length ||= @headers.fetch('content-length', NO_CONTENT_SIZE_PROVIDED)
|
||||
content_length.to_i
|
||||
end
|
||||
|
||||
# Calculates a checksum of given url
|
||||
# @return string
|
||||
def checksum_of(url, etag, last_modified)
|
||||
#noinspection RubyArgCount
|
||||
Zlib::crc32(url + etag + last_modified).to_s
|
||||
end
|
||||
|
||||
# HTTP (Typhoeus) options
|
||||
def http_options
|
||||
{
|
||||
followlocation: true,
|
||||
ssl_verifypeer: false,
|
||||
ssl_verifyhost: 0,
|
||||
timeout: @http_timeout,
|
||||
connecttimeout: @http_connect_timeout
|
||||
}
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,169 @@
|
||||
require 'tempfile'
|
||||
require 'fileutils'
|
||||
require 'json'
|
||||
|
||||
module CartoDB
|
||||
module Datasources
|
||||
# TODO: Error handling, now assumes all done ok
|
||||
class CSVFileDumper
|
||||
|
||||
ORIGINAL_FILE_EXTENSION = '.json'
|
||||
CONVERTED_FILE_EXTENSION = '.csv'
|
||||
HEADERS_FILE_EXTENSION = '_headers.csv'
|
||||
|
||||
FILE_DUMPER_TMP_SUBFOLDER = '/tmp/csv_file_dumper/'
|
||||
|
||||
OUTPUT_ENCODING = 'utf-8'
|
||||
|
||||
def initialize(json_to_csv_conversor, debug_mode = false)
|
||||
@debug_mode = debug_mode
|
||||
@json2csv_conversor = json_to_csv_conversor
|
||||
@temporary_directory = nil
|
||||
@temporary_folder = Time.now.strftime("%Y%m%d_%H%M%S_") + rand(1000).to_s
|
||||
|
||||
@additional_fields = {}
|
||||
|
||||
@files = {}
|
||||
@original_files = {}
|
||||
@headers_file = nil
|
||||
@buffer_size = 8192
|
||||
end
|
||||
|
||||
def buffer_size=(value)
|
||||
@buffer_size = value.to_i if value.to_i > 0
|
||||
end
|
||||
|
||||
def additional_fields=(data = {})
|
||||
@additional_fields = data
|
||||
end
|
||||
|
||||
# This class uses a temporal CSV file per name
|
||||
# optionally dumping also the source JSON in another file if in debug mode
|
||||
# @param name String
|
||||
def begin_dump(name)
|
||||
# Create temp file & open
|
||||
@files[name] = temporary_file(name)
|
||||
@original_files[name] = temporary_file(name, ORIGINAL_FILE_EXTENSION) if @debug_mode
|
||||
@headers_file = temporary_file('', HEADERS_FILE_EXTENSION) if @debug_mode
|
||||
end
|
||||
|
||||
# @param name String
|
||||
# @param data Array
|
||||
# @return Integer number of items dumped
|
||||
def dump(name, data = [])
|
||||
processed_data = @json2csv_conversor.process(data, false, @additional_fields[name]) + "\n"
|
||||
processed_data.encode!(OUTPUT_ENCODING, replace: '')
|
||||
@files[name].write(processed_data)
|
||||
@original_files[name].write(::JSON.dump(data) + "\n") if @debug_mode
|
||||
data.count
|
||||
end
|
||||
|
||||
# @param name String
|
||||
def end_dump(name)
|
||||
if @files[name]
|
||||
@files[name].close
|
||||
end
|
||||
if @original_files[name]
|
||||
@original_files[name].close
|
||||
end
|
||||
end
|
||||
|
||||
# @param names_list Array
|
||||
# @param stream IO
|
||||
def merge_dumps_into_stream(names_list, stream)
|
||||
headers = @json2csv_conversor.generate_headers(@additional_fields[names_list.first]) + "\n"
|
||||
|
||||
streamed_size = headers.length
|
||||
|
||||
stream.write(headers)
|
||||
|
||||
names_list.each do |name|
|
||||
input_stream = File.open(@files[name].path)
|
||||
|
||||
begin
|
||||
buffer = input_stream.read(@buffer_size)
|
||||
if buffer
|
||||
stream.write(buffer)
|
||||
streamed_size += buffer.length
|
||||
end
|
||||
end while buffer
|
||||
|
||||
input_stream.close
|
||||
|
||||
@files[name].unlink unless @debug_mode
|
||||
end
|
||||
|
||||
if @debug_mode && !@headers_file.nil?
|
||||
@headers_file.write(headers)
|
||||
@headers_file.close
|
||||
end
|
||||
|
||||
streamed_size
|
||||
end
|
||||
|
||||
# @param names_list Array
|
||||
# @return String
|
||||
def merge_dumps(names_list = [])
|
||||
headers = @json2csv_conversor.generate_headers(@additional_fields[names_list.first]) + "\n"
|
||||
return_data = headers
|
||||
|
||||
return_data.encode!(OUTPUT_ENCODING, replace: '')
|
||||
|
||||
if @debug_mode && !@headers_file.nil?
|
||||
@headers_file.write(headers)
|
||||
@headers_file.close
|
||||
end
|
||||
|
||||
names_list.each do |name|
|
||||
return_data << File.read(@files[name].path)
|
||||
@files[name].unlink unless @debug_mode
|
||||
end
|
||||
|
||||
# Remove final trailing newline before returning
|
||||
return_data.sub(/\n$/, '')
|
||||
end
|
||||
|
||||
# Return a new temporary file contained inside a tmp subfolder
|
||||
# @param base_name String|nil (optional)
|
||||
def temporary_file(base_name = '', extension = CONVERTED_FILE_EXTENSION)
|
||||
FileUtils.mkdir_p(FILE_DUMPER_TMP_SUBFOLDER) unless File.directory?(FILE_DUMPER_TMP_SUBFOLDER)
|
||||
|
||||
temps_full_path = FILE_DUMPER_TMP_SUBFOLDER + @temporary_folder + '/'
|
||||
FileUtils.mkdir_p(temps_full_path)
|
||||
|
||||
# For the default scenario force encoding, for original files don't touch anything
|
||||
if extension == CONVERTED_FILE_EXTENSION
|
||||
Tempfile.new([base_name.gsub(' ', '_'), extension], temps_full_path, encoding: OUTPUT_ENCODING)
|
||||
else
|
||||
Tempfile.new([base_name.gsub(' ', '_'), extension], temps_full_path)
|
||||
end
|
||||
end
|
||||
|
||||
def file_paths
|
||||
@files.values.map(&:path)
|
||||
end
|
||||
|
||||
def original_file_paths
|
||||
@original_files.values.map(&:path)
|
||||
end
|
||||
|
||||
def headers_path
|
||||
@headers_file.path unless @headers_file.nil?
|
||||
end
|
||||
|
||||
def clean_string(contents)
|
||||
@json2csv_conversor.clean_string(contents)
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
# Intended for tests
|
||||
def destroy_files
|
||||
@files.keys.each { |key| @files[key].close! }
|
||||
@original_files.keys.each { |key| @original_files[key].close! }
|
||||
@headers_file.close! unless @headers_file.nil?
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,79 @@
|
||||
require 'yaml'
|
||||
require_relative '../../../../spec/rspec_configuration'
|
||||
|
||||
require_relative '../../lib/datasources'
|
||||
require_relative '../doubles/user'
|
||||
require 'spec_helper_min'
|
||||
|
||||
include CartoDB::Datasources
|
||||
|
||||
describe DatasourcesFactory do
|
||||
|
||||
def get_config
|
||||
@config ||= YAML.load_file("#{File.dirname(__FILE__)}/../../../../config/app_config.yml")['defaults']
|
||||
end
|
||||
|
||||
describe '#provider_instantiations' do
|
||||
it 'tests all available provider instantiations' do
|
||||
user = FactoryGirl.build(:user)
|
||||
user.stubs('has_feature_flag?').with('gnip_v2').returns(false)
|
||||
DatasourcesFactory.set_config(get_config)
|
||||
|
||||
dropbox_provider = DatasourcesFactory.get_datasource(Url::Dropbox::DATASOURCE_NAME, user)
|
||||
dropbox_provider.is_a?(Url::Dropbox).should eq true
|
||||
|
||||
dropbox_provider = DatasourcesFactory.get_datasource(Url::Box::DATASOURCE_NAME, user)
|
||||
dropbox_provider.is_a?(Url::Box).should eq true
|
||||
|
||||
# Stubs Google Drive client for connectionless testing
|
||||
Google::Apis::DriveV2::DriveService.any_instance.stubs(:get_file)
|
||||
Google::Apis::DriveV2::DriveService.any_instance.stubs(:export_file)
|
||||
Google::Apis::DriveV2::DriveService.any_instance.stubs(:list_files)
|
||||
gdrive_provider = DatasourcesFactory.get_datasource(Url::GDrive::DATASOURCE_NAME, user)
|
||||
gdrive_provider.is_a?(Url::GDrive).should eq true
|
||||
|
||||
url_provider = DatasourcesFactory.get_datasource(Url::PublicUrl::DATASOURCE_NAME, user)
|
||||
url_provider.is_a?(Url::PublicUrl).should eq true
|
||||
|
||||
twitter_provider = DatasourcesFactory.get_datasource(Search::Twitter::DATASOURCE_NAME, user)
|
||||
twitter_provider.is_a?(Search::Twitter).should eq true
|
||||
|
||||
nil_provider = DatasourcesFactory.get_datasource(nil, user)
|
||||
nil_provider.nil?.should eq true
|
||||
|
||||
expect {
|
||||
DatasourcesFactory.get_datasource('blablabla...', user)
|
||||
}.to raise_exception MissingConfigurationError
|
||||
end
|
||||
end
|
||||
|
||||
describe '#customized_config?' do
|
||||
let(:twitter_datasource) { CartoDB::Datasources::Search::Twitter::DATASOURCE_NAME }
|
||||
|
||||
before(:each) do
|
||||
@config = get_config
|
||||
end
|
||||
|
||||
it 'returns false for a random user' do
|
||||
user = FactoryGirl.build(:carto_user, username: 'wadus')
|
||||
DatasourcesFactory.customized_config?(twitter_datasource, user).should be_false
|
||||
end
|
||||
|
||||
it 'returns true for a user with custom config' do
|
||||
user = FactoryGirl.build(:carto_user, username: 'wadus')
|
||||
@config['datasource_search']['twitter_search']['customized_user_list'] = [user.username]
|
||||
DatasourcesFactory.set_config(@config)
|
||||
|
||||
DatasourcesFactory.customized_config?(twitter_datasource, user).should be_true
|
||||
end
|
||||
|
||||
it 'returns true for a user in an organization with custom config' do
|
||||
organization = Carto::Organization.new(name: 'wadus-org')
|
||||
user = FactoryGirl.build(:carto_user, username: 'nowadus', organization: organization)
|
||||
@config['datasource_search']['twitter_search']['customized_orgs_list'] = [organization.name]
|
||||
DatasourcesFactory.set_config(@config)
|
||||
|
||||
DatasourcesFactory.customized_config?(twitter_datasource, user).should be_true
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,35 @@
|
||||
require 'yaml'
|
||||
require_relative '../../lib/datasources'
|
||||
require_relative '../doubles/user'
|
||||
|
||||
include CartoDB::Datasources
|
||||
|
||||
describe Url::Dropbox do
|
||||
|
||||
def get_config
|
||||
@config ||= YAML.load_file("#{File.dirname(__FILE__)}/../../../../config/app_config.yml")['defaults']['oauth']['dropbox']
|
||||
end #get_config
|
||||
|
||||
describe '#manual_test' do
|
||||
it 'with user interaction, tests the full oauth flow and lists files of an account' do
|
||||
user_mock = CartoDB::Datasources::Doubles::User.new
|
||||
|
||||
config = get_config
|
||||
dropbox_datasource = Url::Dropbox.get_new(config, user_mock)
|
||||
|
||||
if config.include?(:access_token)
|
||||
dropbox_datasource.token = config[:access_token]
|
||||
else
|
||||
pending('This test requires manual run, opening the url in a browser, grabbing the code and setting "input" to it')
|
||||
puts dropbox_datasource.get_auth_url
|
||||
input = ''
|
||||
dropbox_datasource.validate_auth_code(input)
|
||||
puts dropbox_datasource.token
|
||||
end
|
||||
data = dropbox_datasource.get_resources_list
|
||||
puts data
|
||||
end
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
@@ -0,0 +1,36 @@
|
||||
require 'yaml'
|
||||
require_relative '../../lib/datasources'
|
||||
require_relative '../doubles/user'
|
||||
|
||||
include CartoDB::Datasources
|
||||
|
||||
describe Url::GDrive do
|
||||
|
||||
def get_config
|
||||
@config ||= YAML.load_file("#{File.dirname(__FILE__)}/../../../../config/app_config.yml")['defaults']['oauth']['gdrive']
|
||||
end
|
||||
|
||||
describe '#manual_test' do
|
||||
it 'with user interaction, tests the full oauth flow and lists files of an account' do
|
||||
config = get_config
|
||||
if !config.include?(:refresh_token)
|
||||
pending('If config unset, this test requires manual running. Check its source code to see what to do')
|
||||
end
|
||||
|
||||
user_mock = CartoDB::Datasources::Doubles::User.new
|
||||
gdrive_datasource = Url::GDrive.get_new(config, user_mock)
|
||||
|
||||
if config.include?(:refresh_token)
|
||||
gdrive_datasource.token = config[:refresh_token]
|
||||
else
|
||||
# Manual testing
|
||||
puts gdrive_datasource.get_auth_url
|
||||
input = ''
|
||||
gdrive_datasource.validate_auth_code(input)
|
||||
puts gdrive_datasource.token
|
||||
end
|
||||
data = gdrive_datasource.get_resources_list
|
||||
puts data
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,36 @@
|
||||
require_relative '../../lib/datasources'
|
||||
require_relative '../../../../spec/helpers/file_server_helper'
|
||||
|
||||
include CartoDB::Datasources
|
||||
include FileServerHelper
|
||||
|
||||
describe Url::PublicUrl do
|
||||
|
||||
describe '#basic_tests' do
|
||||
it 'Some basic download flows of this file provider, including error handling' do
|
||||
url_provider = Url::PublicUrl.get_new
|
||||
|
||||
serve_file 'spec/support/data/cartofante_blue.png' do |url|
|
||||
invalid_url = url + 'invalid'
|
||||
|
||||
data = url_provider.get_resource(url)
|
||||
data.empty?.should eq false
|
||||
expect {
|
||||
url_provider.get_resource(invalid_url)
|
||||
}.to raise_exception DataDownloadError
|
||||
|
||||
url_provider.fetch_headers(url)
|
||||
url_provider.etag_header.empty?.should eq false
|
||||
|
||||
url_provider.fetch_headers(invalid_url).should == {}
|
||||
|
||||
url_provider.etag_header.should be_empty
|
||||
|
||||
url_provider.last_modified_header.should be_empty
|
||||
|
||||
# puts data
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
end
|
||||
@@ -0,0 +1,20 @@
|
||||
module CartoDB
|
||||
module Datasources
|
||||
module Doubles
|
||||
class DataImport
|
||||
|
||||
attr_accessor :id,
|
||||
:service_item_id
|
||||
|
||||
def initialize(attrs = {})
|
||||
@id = attrs.fetch(:id, '123456')
|
||||
@service_item_id = attrs.fetch(:service_item_id, '67890')
|
||||
end
|
||||
|
||||
def save
|
||||
self
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,23 @@
|
||||
module CartoDB
|
||||
module TwitterSearch
|
||||
module Doubles
|
||||
class JSONToCSVConverter
|
||||
|
||||
def initialize(attrs = {})
|
||||
end
|
||||
|
||||
def process(input_data = [], add_headers = false, additional_fields = {})
|
||||
input_data.join("\n")
|
||||
end
|
||||
|
||||
def generate_headers(additional_fields = {})
|
||||
if additional_fields.nil? || additional_fields.empty?
|
||||
''
|
||||
else
|
||||
additional_fields.join(',')
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,18 @@
|
||||
module CartoDB
|
||||
module Datasources
|
||||
module Doubles
|
||||
class Organization
|
||||
|
||||
attr_accessor :twitter_datasource_enabled
|
||||
|
||||
def initialize(attrs = {})
|
||||
@twitter_datasource_enabled = attrs.fetch(:twitter_datasource_enabled, true)
|
||||
end
|
||||
|
||||
def save
|
||||
self
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,26 @@
|
||||
module CartoDB
|
||||
module Datasources
|
||||
module Doubles
|
||||
class SearchTweet
|
||||
|
||||
attr_accessor :user_id,
|
||||
:data_import_id,
|
||||
:service_item_id,
|
||||
:retrieved_items,
|
||||
:state
|
||||
|
||||
def set_importing_state
|
||||
@state = 'importing'
|
||||
end
|
||||
|
||||
def set_complete_state
|
||||
@state = 'complete'
|
||||
end
|
||||
|
||||
def save
|
||||
self
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,45 @@
|
||||
module CartoDB
|
||||
module Datasources
|
||||
module Doubles
|
||||
class User
|
||||
|
||||
attr_accessor :twitter_datasource_enabled,
|
||||
:soft_twitter_datasource_limit,
|
||||
:twitter_datasource_quota,
|
||||
:username,
|
||||
:id
|
||||
|
||||
def initialize(attrs = {})
|
||||
@twitter_datasource_enabled = attrs.fetch(:twitter_datasource_enabled, true)
|
||||
@soft_twitter_datasource_limit = attrs.fetch(:soft_twitter_datasource_limit, false)
|
||||
@twitter_datasource_quota = attrs.fetch(:twitter_datasource_quota, 123)
|
||||
@username = attrs.fetch(:username, 'wadus')
|
||||
@id = attrs.fetch(:id, '000-000')
|
||||
@organization = attrs.fetch(:has_org, false) \
|
||||
? Organization.new({
|
||||
twitter_datasource_enabled: attrs.fetch(:org_twitter_datasource_enabled, true),
|
||||
twitter_datasource_quota: attrs.fetch(:org_twitter_datasource_quota, 123)
|
||||
}) \
|
||||
: nil
|
||||
end
|
||||
|
||||
def organization
|
||||
@organization
|
||||
end
|
||||
|
||||
def save
|
||||
self
|
||||
end
|
||||
|
||||
def remaining_twitter_quota
|
||||
if @organization.nil?
|
||||
@twitter_datasource_quota
|
||||
else
|
||||
@organization.twitter_datasource_quota
|
||||
end
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,63 @@
|
||||
{
|
||||
"displayFieldName": "NAME",
|
||||
"fieldAliases": {
|
||||
"OBJECTID": "OBJECTID",
|
||||
"WDPAID": "WDPAID",
|
||||
"NAME": "NAME"
|
||||
},
|
||||
"geometryType": "esriGeometryPolygon",
|
||||
"spatialReference": {
|
||||
"wkid": 4326,
|
||||
"latestWkid": 4326
|
||||
},
|
||||
"fields": [
|
||||
{
|
||||
"name": "OBJECTID",
|
||||
"type": "esriFieldTypeOID",
|
||||
"alias": "OBJECTID"
|
||||
},
|
||||
{
|
||||
"name": "WDPAID",
|
||||
"type": "esriFieldTypeInteger",
|
||||
"alias": "WDPAID"
|
||||
},
|
||||
{
|
||||
"name": "NAME",
|
||||
"type": "esriFieldTypeString",
|
||||
"alias": "NAME",
|
||||
"length": 254
|
||||
}
|
||||
],
|
||||
"features": [
|
||||
{
|
||||
"attributes": {
|
||||
"OBJECTID": 1,
|
||||
"WDPAID": 991,
|
||||
"NAME": "Name of object 1"
|
||||
},
|
||||
"geometry": {
|
||||
"fake": "geom"
|
||||
}
|
||||
},
|
||||
{
|
||||
"attributes": {
|
||||
"OBJECTID": 2,
|
||||
"WDPAID": 992,
|
||||
"NAME": "Name of object 2"
|
||||
},
|
||||
"geometry": {
|
||||
"fake": "geom"
|
||||
}
|
||||
},
|
||||
{
|
||||
"attributes": {
|
||||
"OBJECTID": 3,
|
||||
"WDPAID": 993,
|
||||
"NAME": "Name of object 3"
|
||||
},
|
||||
"geometry": {
|
||||
"fake": "geom"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1 @@
|
||||
{"objectIdFieldName":"OBJECTID","objectIds":[1,2,3,4,5,6,7,8,9,10]}
|
||||
@@ -0,0 +1 @@
|
||||
{"objectIdFieldName":"OBJECTID","objectIds":[1,2,3]}
|
||||
@@ -0,0 +1,3 @@
|
||||
{"layers":[
|
||||
{"id":0, "name":"first layer", "type":"Feature Layer"}
|
||||
]}
|
||||
@@ -0,0 +1,22 @@
|
||||
{"currentVersion":10.22,
|
||||
"id":0,
|
||||
"name":"Test Feature",
|
||||
"type":"Feature Layer",
|
||||
"description":"Sample metadata payload",
|
||||
"geometryType":"esriGeometryPolygon",
|
||||
"copyrightText":"CartoDB",
|
||||
"fields":[
|
||||
{"name":"OBJECTID",
|
||||
"type":"esriFieldTypeOID",
|
||||
"alias":"OBJECTID",
|
||||
"domain":null},
|
||||
{"name":"NAME",
|
||||
"type":"esriFieldTypeString",
|
||||
"alias":"NAME",
|
||||
"length":254,
|
||||
"domain":null}
|
||||
],
|
||||
"maxRecordCount":1000,
|
||||
"supportsAdvancedQueries":true,
|
||||
"supportedQueryFormats":"JSON,AMF",
|
||||
"useStandardizedQueries":true}
|
||||
@@ -0,0 +1 @@
|
||||
{"objectIdFieldName":"OBJECTID","objectIds":[7,6,10,4,5,2,1,8,9,3]}
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,2 @@
|
||||
"link","body","objectType","postedTime","favoritesCount","twitter_lang","retweetCount","actor_id","actor_displayName","actor_image","actor_summary","actor_postedTime","actor_location","actor_utcOffset","actor_preferredUsername","actor_friendsCount","actor_followersCount","actor_listedCount","actor_statusesCount","actor_verified","inReplyTo_link","geo","twitter_entities","location_geo","location_name","the_geom","category_name","category_terms"
|
||||
"http://twitter.com/charley_glynn/statuses/834008296361172992","RT @uSIG_CCHS_CSIC: Carto tips: Using blend modes and opacity levels-https://t.co/eGDWGkfWCv via @OrdnanceSurvey https://t.co/qEfFQpLGvL","activity","2017-02-21T11:54:07.000Z","0","en","6","id:twitter.com:492926788","Charley Glynn","https://pbs.twimg.com/profile_images/740123846951460864/gi1ee2lJ_normal.jpg","happy husband, doting dad, Everton fan, music, film & design fan, pushing pixels and joining dots @ordnancesurvey working on #GeoDataViz","2012-02-15T08:23:02.000Z","{""objectType"":""place"",""displayName"":""Southampton, England""}","0","charley_glynn","2482","1415","168","7536","false",,,"{""hashtags"":[],""urls"":[{""url"":""https://t.co/eGDWGkfWCv"",""expanded_url"":""https://www.ordnancesurvey.co.uk/blog/2017/02/carto-tips-using-blend-modes-opacity-levels/"",""display_url"":""ordnancesurvey.co.uk/blog/2017/02/c…"",""indices"":[69,92]}],""user_mentions"":[{""screen_name"":""uSIG_CCHS_CSIC"",""name"":""uSIG (CCHS-CSIC)"",""id"":704613732614344704,""id_str"":""704613732614344704"",""indices"":[3,18]},{""screen_name"":""OrdnanceSurvey"",""name"":""Ordnance Survey"",""id"":22614266,""id_str"":""22614266"",""indices"":[97,112]}],""symbols"":[],""media"":[{""id"":833943995302760449,""id_str"":""833943995302760449"",""indices"":[113,136],""media_url"":""http://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg"",""media_url_https"":""https://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg"",""url"":""https://t.co/qEfFQpLGvL"",""display_url"":""pic.twitter.com/qEfFQpLGvL"",""expanded_url"":""https://twitter.com/uSIG_CCHS_CSIC/status/833972033960620032/photo/1"",""type"":""photo"",""sizes"":{""medium"":{""w"":600,""h"":415,""resize"":""fit""},""large"":{""w"":650,""h"":450,""resize"":""fit""},""thumb"":{""w"":150,""h"":150,""resize"":""crop""},""small"":{""w"":340,""h"":235,""resize"":""fit""}},""source_status_id"":833972033960620032,""source_status_id_str"":""833972033960620032"",""source_user_id"":704613732614344704,""source_user_id_str"":""704613732614344704""}]}",,,"{""coordinates"":[-1.40428,50.90395],""type"":""point""}","1","carto"
|
||||
|
@@ -0,0 +1,409 @@
|
||||
{
|
||||
"results": [
|
||||
{
|
||||
"id": "tag:search.twitter.com,2005:834008296361172992",
|
||||
"objectType": "activity",
|
||||
"verb": "share",
|
||||
"postedTime": "2017-02-21T11:54:07.000Z",
|
||||
"generator": {
|
||||
"displayName": "Twitter Web Client",
|
||||
"link": "http://twitter.com"
|
||||
},
|
||||
"provider": {
|
||||
"objectType": "service",
|
||||
"displayName": "Twitter",
|
||||
"link": "http://www.twitter.com"
|
||||
},
|
||||
"link": "http://twitter.com/charley_glynn/statuses/834008296361172992",
|
||||
"body": "RT @uSIG_CCHS_CSIC: Carto tips: Using blend modes and opacity levels-https://t.co/eGDWGkfWCv via @OrdnanceSurvey https://t.co/qEfFQpLGvL",
|
||||
"actor": {
|
||||
"objectType": "person",
|
||||
"id": "id:twitter.com:492926788",
|
||||
"link": "http://www.twitter.com/charley_glynn",
|
||||
"displayName": "Charley Glynn",
|
||||
"postedTime": "2012-02-15T08:23:02.000Z",
|
||||
"image": "https://pbs.twimg.com/profile_images/740123846951460864/gi1ee2lJ_normal.jpg",
|
||||
"summary": "happy husband, doting dad, Everton fan, music, film & design fan, pushing pixels and joining dots @ordnancesurvey working on #GeoDataViz",
|
||||
"friendsCount": 2482,
|
||||
"followersCount": 1415,
|
||||
"listedCount": 168,
|
||||
"statusesCount": 7536,
|
||||
"twitterTimeZone": "London",
|
||||
"verified": false,
|
||||
"utcOffset": "0",
|
||||
"preferredUsername": "charley_glynn",
|
||||
"languages": [
|
||||
"en"
|
||||
],
|
||||
"links": [
|
||||
{
|
||||
"href": "http://www.cartoblography.wordpress.com",
|
||||
"rel": "me"
|
||||
}
|
||||
],
|
||||
"location": {
|
||||
"objectType": "place",
|
||||
"displayName": "Southampton, England"
|
||||
},
|
||||
"favoritesCount": 7414
|
||||
},
|
||||
"object": {
|
||||
"id": "tag:search.twitter.com,2005:833972033960620032",
|
||||
"objectType": "activity",
|
||||
"verb": "post",
|
||||
"postedTime": "2017-02-21T09:30:02.000Z",
|
||||
"generator": {
|
||||
"displayName": "TweetDeck",
|
||||
"link": "https://about.twitter.com/products/tweetdeck"
|
||||
},
|
||||
"provider": {
|
||||
"objectType": "service",
|
||||
"displayName": "Twitter",
|
||||
"link": "http://www.twitter.com"
|
||||
},
|
||||
"link": "http://twitter.com/uSIG_CCHS_CSIC/statuses/833972033960620032",
|
||||
"body": "Carto tips: Using blend modes and opacity levels-https://t.co/eGDWGkfWCv via @OrdnanceSurvey https://t.co/qEfFQpLGvL",
|
||||
"display_text_range": [
|
||||
0,
|
||||
92
|
||||
],
|
||||
"actor": {
|
||||
"objectType": "person",
|
||||
"id": "id:twitter.com:704613732614344704",
|
||||
"link": "http://www.twitter.com/uSIG_CCHS_CSIC",
|
||||
"displayName": "uSIG (CCHS-CSIC)",
|
||||
"postedTime": "2016-03-01T10:26:19.759Z",
|
||||
"image": "https://pbs.twimg.com/profile_images/802141753553854464/4ZSz4PI-_normal.jpg",
|
||||
"summary": "Unidad SIG. Espacio, tiempo, ciencia y tecnología #SIG #Cartografía y #Teledetección en Ciencias Humanas y Sociales @CSIC. #GIS #Cartography #RemoteSensing",
|
||||
"friendsCount": 317,
|
||||
"followersCount": 515,
|
||||
"listedCount": 41,
|
||||
"statusesCount": 1462,
|
||||
"twitterTimeZone": null,
|
||||
"verified": false,
|
||||
"utcOffset": null,
|
||||
"preferredUsername": "uSIG_CCHS_CSIC",
|
||||
"languages": [
|
||||
"es"
|
||||
],
|
||||
"links": [
|
||||
{
|
||||
"href": "http://unidadsig.cchs.csic.es/sig/",
|
||||
"rel": "me"
|
||||
}
|
||||
],
|
||||
"location": {
|
||||
"objectType": "place",
|
||||
"displayName": "Madrid, España"
|
||||
},
|
||||
"favoritesCount": 811
|
||||
},
|
||||
"object": {
|
||||
"objectType": "note",
|
||||
"id": "object:search.twitter.com,2005:833972033960620032",
|
||||
"summary": "Carto tips: Using blend modes and opacity levels-https://t.co/eGDWGkfWCv via @OrdnanceSurvey https://t.co/qEfFQpLGvL",
|
||||
"link": "http://twitter.com/uSIG_CCHS_CSIC/statuses/833972033960620032",
|
||||
"postedTime": "2017-02-21T09:30:02.000Z"
|
||||
},
|
||||
"favoritesCount": 11,
|
||||
"twitter_entities": {
|
||||
"hashtags": [],
|
||||
"urls": [
|
||||
{
|
||||
"url": "https://t.co/eGDWGkfWCv",
|
||||
"expanded_url": "https://www.ordnancesurvey.co.uk/blog/2017/02/carto-tips-using-blend-modes-opacity-levels/",
|
||||
"display_url": "ordnancesurvey.co.uk/blog/2017/02/c…",
|
||||
"indices": [
|
||||
49,
|
||||
72
|
||||
]
|
||||
}
|
||||
],
|
||||
"user_mentions": [
|
||||
{
|
||||
"screen_name": "OrdnanceSurvey",
|
||||
"name": "Ordnance Survey",
|
||||
"id": 22614266,
|
||||
"id_str": "22614266",
|
||||
"indices": [
|
||||
77,
|
||||
92
|
||||
]
|
||||
}
|
||||
],
|
||||
"symbols": [],
|
||||
"media": [
|
||||
{
|
||||
"id": 833943995302760449,
|
||||
"id_str": "833943995302760449",
|
||||
"indices": [
|
||||
93,
|
||||
116
|
||||
],
|
||||
"media_url": "http://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg",
|
||||
"media_url_https": "https://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg",
|
||||
"url": "https://t.co/qEfFQpLGvL",
|
||||
"display_url": "pic.twitter.com/qEfFQpLGvL",
|
||||
"expanded_url": "https://twitter.com/uSIG_CCHS_CSIC/status/833972033960620032/photo/1",
|
||||
"type": "photo",
|
||||
"sizes": {
|
||||
"medium": {
|
||||
"w": 600,
|
||||
"h": 415,
|
||||
"resize": "fit"
|
||||
},
|
||||
"large": {
|
||||
"w": 650,
|
||||
"h": 450,
|
||||
"resize": "fit"
|
||||
},
|
||||
"thumb": {
|
||||
"w": 150,
|
||||
"h": 150,
|
||||
"resize": "crop"
|
||||
},
|
||||
"small": {
|
||||
"w": 340,
|
||||
"h": 235,
|
||||
"resize": "fit"
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
},
|
||||
"twitter_extended_entities": {
|
||||
"media": [
|
||||
{
|
||||
"id": 833943995302760449,
|
||||
"id_str": "833943995302760449",
|
||||
"indices": [
|
||||
93,
|
||||
116
|
||||
],
|
||||
"media_url": "http://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg",
|
||||
"media_url_https": "https://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg",
|
||||
"url": "https://t.co/qEfFQpLGvL",
|
||||
"display_url": "pic.twitter.com/qEfFQpLGvL",
|
||||
"expanded_url": "https://twitter.com/uSIG_CCHS_CSIC/status/833972033960620032/photo/1",
|
||||
"type": "animated_gif",
|
||||
"sizes": {
|
||||
"medium": {
|
||||
"w": 600,
|
||||
"h": 415,
|
||||
"resize": "fit"
|
||||
},
|
||||
"large": {
|
||||
"w": 650,
|
||||
"h": 450,
|
||||
"resize": "fit"
|
||||
},
|
||||
"thumb": {
|
||||
"w": 150,
|
||||
"h": 150,
|
||||
"resize": "crop"
|
||||
},
|
||||
"small": {
|
||||
"w": 340,
|
||||
"h": 235,
|
||||
"resize": "fit"
|
||||
}
|
||||
},
|
||||
"video_info": {
|
||||
"aspect_ratio": [
|
||||
13,
|
||||
9
|
||||
],
|
||||
"variants": [
|
||||
{
|
||||
"bitrate": 0,
|
||||
"content_type": "video/mp4",
|
||||
"url": "https://video.twimg.com/tweet_video/C5LDpTKXUAE1xex.mp4"
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
]
|
||||
},
|
||||
"twitter_lang": "en",
|
||||
"twitter_filter_level": "low"
|
||||
},
|
||||
"favoritesCount": 0,
|
||||
"twitter_entities": {
|
||||
"hashtags": [],
|
||||
"urls": [
|
||||
{
|
||||
"url": "https://t.co/eGDWGkfWCv",
|
||||
"expanded_url": "https://www.ordnancesurvey.co.uk/blog/2017/02/carto-tips-using-blend-modes-opacity-levels/",
|
||||
"display_url": "ordnancesurvey.co.uk/blog/2017/02/c…",
|
||||
"indices": [
|
||||
69,
|
||||
92
|
||||
]
|
||||
}
|
||||
],
|
||||
"user_mentions": [
|
||||
{
|
||||
"screen_name": "uSIG_CCHS_CSIC",
|
||||
"name": "uSIG (CCHS-CSIC)",
|
||||
"id": 704613732614344704,
|
||||
"id_str": "704613732614344704",
|
||||
"indices": [
|
||||
3,
|
||||
18
|
||||
]
|
||||
},
|
||||
{
|
||||
"screen_name": "OrdnanceSurvey",
|
||||
"name": "Ordnance Survey",
|
||||
"id": 22614266,
|
||||
"id_str": "22614266",
|
||||
"indices": [
|
||||
97,
|
||||
112
|
||||
]
|
||||
}
|
||||
],
|
||||
"symbols": [],
|
||||
"media": [
|
||||
{
|
||||
"id": 833943995302760449,
|
||||
"id_str": "833943995302760449",
|
||||
"indices": [
|
||||
113,
|
||||
136
|
||||
],
|
||||
"media_url": "http://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg",
|
||||
"media_url_https": "https://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg",
|
||||
"url": "https://t.co/qEfFQpLGvL",
|
||||
"display_url": "pic.twitter.com/qEfFQpLGvL",
|
||||
"expanded_url": "https://twitter.com/uSIG_CCHS_CSIC/status/833972033960620032/photo/1",
|
||||
"type": "photo",
|
||||
"sizes": {
|
||||
"medium": {
|
||||
"w": 600,
|
||||
"h": 415,
|
||||
"resize": "fit"
|
||||
},
|
||||
"large": {
|
||||
"w": 650,
|
||||
"h": 450,
|
||||
"resize": "fit"
|
||||
},
|
||||
"thumb": {
|
||||
"w": 150,
|
||||
"h": 150,
|
||||
"resize": "crop"
|
||||
},
|
||||
"small": {
|
||||
"w": 340,
|
||||
"h": 235,
|
||||
"resize": "fit"
|
||||
}
|
||||
},
|
||||
"source_status_id": 833972033960620032,
|
||||
"source_status_id_str": "833972033960620032",
|
||||
"source_user_id": 704613732614344704,
|
||||
"source_user_id_str": "704613732614344704"
|
||||
}
|
||||
]
|
||||
},
|
||||
"twitter_extended_entities": {
|
||||
"media": [
|
||||
{
|
||||
"id": 833943995302760449,
|
||||
"id_str": "833943995302760449",
|
||||
"indices": [
|
||||
113,
|
||||
136
|
||||
],
|
||||
"media_url": "http://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg",
|
||||
"media_url_https": "https://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg",
|
||||
"url": "https://t.co/qEfFQpLGvL",
|
||||
"display_url": "pic.twitter.com/qEfFQpLGvL",
|
||||
"expanded_url": "https://twitter.com/uSIG_CCHS_CSIC/status/833972033960620032/photo/1",
|
||||
"type": "animated_gif",
|
||||
"sizes": {
|
||||
"medium": {
|
||||
"w": 600,
|
||||
"h": 415,
|
||||
"resize": "fit"
|
||||
},
|
||||
"large": {
|
||||
"w": 650,
|
||||
"h": 450,
|
||||
"resize": "fit"
|
||||
},
|
||||
"thumb": {
|
||||
"w": 150,
|
||||
"h": 150,
|
||||
"resize": "crop"
|
||||
},
|
||||
"small": {
|
||||
"w": 340,
|
||||
"h": 235,
|
||||
"resize": "fit"
|
||||
}
|
||||
},
|
||||
"source_status_id": 833972033960620032,
|
||||
"source_status_id_str": "833972033960620032",
|
||||
"source_user_id": 704613732614344704,
|
||||
"source_user_id_str": "704613732614344704",
|
||||
"video_info": {
|
||||
"aspect_ratio": [
|
||||
13,
|
||||
9
|
||||
],
|
||||
"variants": [
|
||||
{
|
||||
"bitrate": 0,
|
||||
"content_type": "video/mp4",
|
||||
"url": "https://video.twimg.com/tweet_video/C5LDpTKXUAE1xex.mp4"
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
]
|
||||
},
|
||||
"twitter_lang": "en",
|
||||
"retweetCount": 6,
|
||||
"gnip": {
|
||||
"profileLocations": [
|
||||
{
|
||||
"address": {
|
||||
"country": "United Kingdom",
|
||||
"countryCode": "GB",
|
||||
"locality": "Southampton",
|
||||
"region": "England",
|
||||
"subRegion": "City of Southampton"
|
||||
},
|
||||
"displayName": "Southampton, England, United Kingdom",
|
||||
"geo": {
|
||||
"coordinates": [
|
||||
-1.40428,
|
||||
50.90395
|
||||
],
|
||||
"type": "point"
|
||||
},
|
||||
"objectType": "place"
|
||||
}
|
||||
],
|
||||
"matching_rules": [
|
||||
{
|
||||
"value": "(carto) (has:geo OR has:profile_geo)",
|
||||
"tag": null
|
||||
}
|
||||
],
|
||||
"urls": [
|
||||
{
|
||||
"url": "https://t.co/eGDWGkfWCv",
|
||||
"expanded_url": "https://www.ordnancesurvey.co.uk/blog/2017/02/carto-tips-using-blend-modes-opacity-levels/",
|
||||
"expanded_status": 200,
|
||||
"expanded_url_title": "Carto tips: Using blend modes and opacity levels - Ordnance Survey Blog",
|
||||
"expanded_url_description": "Colour is one of the main graphic elements that a cartographer uses to make their map clear to read. Amongst other things we use colour to create familiarity, to differentiate features and to create a clear visual hierarchy. There are many things we can do to the features on our maps to change their appearance... Read More"
|
||||
}
|
||||
]
|
||||
},
|
||||
"twitter_filter_level": "low"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1 @@
|
||||
stream_input_1
|
||||
@@ -0,0 +1 @@
|
||||
stream_input_2
|
||||
@@ -0,0 +1,40 @@
|
||||
"id","verb","link","body","objectType","postedTime","favoritesCount","twitter_filter_level","twitter_lang","retweetCount","actor_objectType","actor_id","actor_link","actor_displayName","actor_image","actor_summary","actor_postedTime","actor_links","actor_location","actor_utcOffset","actor_preferredUsername","actor_languages","actor_twitterTimeZone","actor_friendsCount","actor_followersCount","actor_listedCount","actor_statusesCount","actor_verified","generator_displayName","generator_link","provider_objectType","provider_displayName","provider_link","inReplyTo_link","geo","twitter_entities","object_objectType","object_id","object_summary","object_postedTime","object_link","location_objectType","location_displayName","location_link","location_geo","location_streetAddress","location_name","gnip","the_geom","category_name","category_terms"
|
||||
"tag:search.twitter.com,2005:496274441584668672","post","http://twitter.com/erictheise/statuses/496274441584668672","sad to be the guy that knocked ""locahost"" out of @cartoDB's docs for #Odyssey.js; i've certainly had that sentiment towards my dev machine.","activity","2014-08-04T12:40:21.000Z","0","medium","en","0","person","id:twitter.com:94817737","http://www.twitter.com/erictheise","Eric Theise","https://pbs.twimg.com/profile_images/1994194116/Theise_normal.jpg","Software engineer; photographer, experimental filmmaker, cartographer, geographer, vocalizer, yoga student, reader & writer, eater & drinker.","2009-12-05T16:06:12.000Z","[{""href"":null,""rel"":""me""}]","{""objectType"":""place"",""displayName"":""San Francisco""}","-25200","erictheise","[""en""]","Pacific Time (US & Canada)","698","306","25","1322","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[{""text"":""Odyssey"",""indices"":[69,77]}],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[49,57]}],""symbols"":[]}","note","object:search.twitter.com,2005:496274441584668672","sad to be the guy that knocked ""locahost"" out of @cartoDB's docs for #Odyssey.js; i've certainly had that sentiment towards my dev machine.","2014-08-04T12:40:21.000Z","http://twitter.com/erictheise/statuses/496274441584668672","place","San Francisco, CA","https://api.twitter.com/1.1/geo/id/5a110d312052166f.json","{""type"":""Polygon"",""coordinates"":[[[-122.514926,37.708075],[-122.514926,37.833238],[-122.328001,37.833238],[-122.328001,37.708075]]]}",,"San Francisco","{""klout_score"":41,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-122.514926,37.708075],[-122.514926,37.833238],[-122.328001,37.833238],[-122.328001,37.708075]]]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:495565956908064769","post","http://twitter.com/Jmholleran/statuses/495565956908064769","FrackMap #map http://t.co/BLWl3zYKTn","activity","2014-08-02T13:45:05.000Z","0","medium","en","0","person","id:twitter.com:27128139","http://www.twitter.com/Jmholleran","John Munro Holleran","https://pbs.twimg.com/profile_images/460662090819051520/lzLtBzUb_normal.png","Modest! intelligent, articulate & generally just amazing! Trade Unionist, learning facilitator 'n that","2009-03-27T23:36:39.000Z","[{""href"":null,""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Glasgow/Scotland""}",,"Jmholleran","[""en""]",,"411","192","1","1643","false","Twitter for Android","http://twitter.com/download/android","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[55.8641,-4.282299]}","{""hashtags"":[{""text"":""map"",""indices"":[9,13]}],""trends"":[],""urls"":[{""url"":""http://t.co/BLWl3zYKTn"",""expanded_url"":""http://energydesk.cartodb.com/viz/18162194-ecbd-11e3-aba1-0e230854a1cb/embed_map?title=true&description=true&search=false&shareable=true&cartodb_logo=true&layer_selector=false&legends=true&scrollwheel=true&fullscreen=true&sublayer_options=1%7C1%7C1%7C1&sql=&sw_lat=49.97948776108648&sw_lon=-12.7001953125&ne_lat=56.80087831233043&ne_lon=8.96484375"",""display_url"":""energydesk.cartodb.com/viz/18162194-e…"",""indices"":[14,36]}],""user_mentions"":[],""symbols"":[]}","note","object:search.twitter.com,2005:495565956908064769","FrackMap #map http://t.co/BLWl3zYKTn","2014-08-02T13:45:05.000Z","http://twitter.com/Jmholleran/statuses/495565956908064769","place","Glasgow","https://api.twitter.com/1.1/geo/id/791e00bcadc4615f.json","{""type"":""Polygon"",""coordinates"":[[[-4.3932845,55.796184],[-4.3932845,55.9204214],[-4.0902182,55.9204214],[-4.0902182,55.796184]]]}",,"Glasgow","{""urls"":[{""url"":""http://t.co/BLWl3zYKTn"",""expanded_url"":""http://energydesk.cartodb.com/viz/18162194-ecbd-11e3-aba1-0e230854a1cb/embed_map?title=true&description=true&search=false&shareable=true&cartodb_logo=true&layer_selector=false&legends=true&scrollwheel=true&fullscreen=true&sublayer_options=1%7C1%7C1%7C1&sql&sw_lat=49.97948776108648&sw_lon=-12.7001953125&ne_lat=56.80087831233043&ne_lon=8.96484375"",""expanded_status"":200}],""klout_score"":23}","{""type"":""Point"",""coordinates"":[-4.282299,55.8641]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:494521462682685440","post","http://twitter.com/SonOfJorEl/statuses/494521462682685440","@iriberri1 @cartoDB you are saying it is not supported, but you have an entire blog post about said feature : http://t.co/gvuUWv9sge","activity","2014-07-30T16:34:39.000Z","0","medium","en","0","person","id:twitter.com:15628310","http://www.twitter.com/SonOfJorEl","Joey Schluchter","https://pbs.twimg.com/profile_images/466697750013562880/ngwKpO1j_normal.jpeg","Recovering golfer. Helping feed the world to software. Still love persimmon.","2008-07-28T05:34:03.000Z","[{""href"":""http://www.exzeo.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Clearwater, FL""}","-14400","SonOfJorEl","[""en""]","Eastern Time (US & Canada)","368","2053","5","2143","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com","http://twitter.com/iriberri1/statuses/494488277765091328",,"{""hashtags"":[],""symbols"":[],""urls"":[{""url"":""http://t.co/gvuUWv9sge"",""expanded_url"":""http://blog.cartodb.com/got-files-weve-got-a-import-api/"",""display_url"":""blog.cartodb.com/got-files-weve…"",""indices"":[110,132]}],""user_mentions"":[{""screen_name"":""iriberri1"",""name"":""Carla"",""id"":102197411,""id_str"":""102197411"",""indices"":[0,10]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[11,19]}]}","note","object:search.twitter.com,2005:494521462682685440","@iriberri1 @cartoDB you are saying it is not supported, but you have an entire blog post about said feature : http://t.co/gvuUWv9sge","2014-07-30T16:34:39.000Z","http://twitter.com/SonOfJorEl/statuses/494521462682685440","place","Tampa, FL","https://api.twitter.com/1.1/geo/id/dc62519fda13b4ec.json","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}",,"Tampa","{""urls"":[{""url"":""http://t.co/gvuUWv9sge"",""expanded_url"":""http://blog.cartodb.com/got-files-weve-got-a-import-api/"",""expanded_status"":200}],""klout_score"":42,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:494511938563342336","post","http://twitter.com/xavijam/statuses/494511938563342336",". @Javier is giving birth last @cartodb feature. All I can say is … GIF over maps! http://t.co/oC3DEK5cP6","activity","2014-07-30T15:56:48.000Z","0","medium","en","0","person","id:twitter.com:26775253","http://www.twitter.com/xavijam","Javier Álvarez","https://pbs.twimg.com/profile_images/3025038956/a2919d353eb2b1a22756d7ac79847480_normal.jpeg","I love gentoo penguins","2009-03-26T15:36:52.000Z","[{""href"":""http://xavij.am"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Antarctic Peninsula""}","7200","xavijam","[""en""]","Madrid","250","395","26","4981","false","Tweetbot for iΟS","http://tapbots.com/tweetbot","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""javier"",""name"":""Javier Arce"",""id"":39083,""id_str"":""39083"",""indices"":[2,9]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[31,39]}],""symbols"":[],""media"":[{""id"":494511934629093378,""id_str"":""494511934629093378"",""indices"":[83,105],""media_url"":""http://pbs.twimg.com/media/Btzb_AjCIAI0cLr.jpg"",""media_url_https"":""https://pbs.twimg.com/media/Btzb_AjCIAI0cLr.jpg"",""url"":""http://t.co/oC3DEK5cP6"",""display_url"":""pic.twitter.com/oC3DEK5cP6"",""expanded_url"":""http://twitter.com/xavijam/status/494511938563342336/photo/1"",""type"":""photo"",""sizes"":{""small"":{""w"":340,""h"":255,""resize"":""fit""},""medium"":{""w"":600,""h"":450,""resize"":""fit""},""large"":{""w"":1024,""h"":768,""resize"":""fit""},""thumb"":{""w"":150,""h"":150,""resize"":""crop""}}}]}","note","object:search.twitter.com,2005:494511938563342336",". @Javier is giving birth last @cartodb feature. All I can say is … GIF over maps! http://t.co/oC3DEK5cP6","2014-07-30T15:56:48.000Z","http://twitter.com/xavijam/statuses/494511938563342336","place","Trafalgar","https://api.twitter.com/1.1/geo/id/0144b1172069289c.json","{""type"":""Polygon"",""coordinates"":[[[-3.70582,40.4275708],[-3.70582,40.4386747],[-3.6956326,40.4386747],[-3.6956326,40.4275708]]]}",,"Trafalgar","{""urls"":[{""url"":""http://t.co/oC3DEK5cP6"",""expanded_url"":""http://twitter.com/xavijam/status/494511938563342336/photo/1"",""expanded_status"":200}],""klout_score"":39,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-3.70582,40.4275708],[-3.70582,40.4386747],[-3.6956326,40.4386747],[-3.6956326,40.4275708]]]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:494485754848894978","post","http://twitter.com/SonOfJorEl/statuses/494485754848894978","@cartoDB I’ve emailed support (which you claim has a 24hr turnaround) with minimal response. I am a paying customer ($149 mo). For how long?","activity","2014-07-30T14:12:45.000Z","0","medium","en","0","person","id:twitter.com:15628310","http://www.twitter.com/SonOfJorEl","Joey Schluchter","https://pbs.twimg.com/profile_images/466697750013562880/ngwKpO1j_normal.jpeg","Recovering golfer. Helping feed the world to software. Still love persimmon.","2008-07-28T05:34:03.000Z","[{""href"":""http://www.exzeo.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Clearwater, FL""}","-14400","SonOfJorEl","[""en""]","Eastern Time (US & Canada)","368","2053","5","2142","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[0,8]}],""symbols"":[]}","note","object:search.twitter.com,2005:494485754848894978","@cartoDB I’ve emailed support (which you claim has a 24hr turnaround) with minimal response. I am a paying customer ($149 mo). For how long?","2014-07-30T14:12:45.000Z","http://twitter.com/SonOfJorEl/statuses/494485754848894978","place","Tampa, FL","https://api.twitter.com/1.1/geo/id/dc62519fda13b4ec.json","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}",,"Tampa","{""klout_score"":42,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:494276921765543936","post","http://twitter.com/httsan/statuses/494276921765543936","""@gisuser: Online Mapping for Beginners – The Map Academy from CartoDB: TweetSomething innovative from… http://t.co/Zhh882TV2F""","activity","2014-07-30T00:22:56.000Z","0","medium","en","0","person","id:twitter.com:28471178","http://www.twitter.com/httsan","hartanto","https://pbs.twimg.com/profile_images/2217327231/IMG0033_normal.jpg","Lahir di Indonesia | Nguli di @NEOnetBPPT @BPPTeknologi | Ngobrol di @RSGISForum @PadangZaitun | Nyantri di @MichiganStateU","2009-04-03T01:32:35.000Z","[{""href"":""http://about.me/httsan"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""East Lansing""}","-14400","httsan","[""en""]","Eastern Time (US & Canada)","976","873","12","25725","false","Twitter for Android","http://twitter.com/download/android","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[42.7539688,-84.4221123]}","{""hashtags"":[],""symbols"":[],""urls"":[{""url"":""http://t.co/Zhh882TV2F"",""expanded_url"":""http://goo.gl/fb/xVB5rr"",""display_url"":""goo.gl/fb/xVB5rr"",""indices"":[104,126]}],""user_mentions"":[{""screen_name"":""gisuser"",""name"":""GISuser GeoTech News"",""id"":16958615,""id_str"":""16958615"",""indices"":[1,9]}]}","note","object:search.twitter.com,2005:494276921765543936","""@gisuser: Online Mapping for Beginners – The Map Academy from CartoDB: TweetSomething innovative from… http://t.co/Zhh882TV2F""","2014-07-30T00:22:56.000Z","http://twitter.com/httsan/statuses/494276921765543936","place","Haslett, MI","https://api.twitter.com/1.1/geo/id/4399b5004a3b4d9a.json","{""type"":""Polygon"",""coordinates"":[[[-84.447506,42.731229],[-84.447506,42.769688],[-84.363432,42.769688],[-84.363432,42.731229]]]}",,"Haslett","{""urls"":[{""url"":""http://t.co/Zhh882TV2F"",""expanded_url"":""http://blog.gisuser.com/2014/07/29/online-mapping-for-beginners-the-map-academy-from-cartodb/?utm_source=feedburner&utm_medium=twitter&utm_campaign=Feed:%20gisuser%20(GISUser.com%20-%20GIS,%20Mapping,%20Geospatial,%20and%20location%20technology%20news)"",""expanded_status"":200}],""klout_score"":51,""language"":{""value"":""en""}}","{""type"":""Point"",""coordinates"":[-84.4221123,42.7539688]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:494274906528288769","post","http://twitter.com/carygeo/statuses/494274906528288769","Attention mappers: iframe, #javascript, app development examples from the #CartoDB API/team! Well done @andrewxhill. http://t.co/cJM5fzgoyL","activity","2014-07-30T00:14:55.000Z","0","medium","en","0","person","id:twitter.com:2641688154","http://www.twitter.com/carygeo","Cary Greenwood","https://pbs.twimg.com/profile_images/488501325697544192/gFMxHuLV_normal.jpeg","geography, maps, data visualization, ui/ux design","2014-07-14T01:35:22.000Z","[{""href"":null,""rel"":""me""}]",,,"carygeo","[""en""]",,"109","46","0","47","false","Twitter for iPhone","http://twitter.com/download/iphone","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[34.13846767,-118.3620715]}","{""hashtags"":[{""text"":""javascript"",""indices"":[27,38]},{""text"":""CartoDB"",""indices"":[74,82]}],""symbols"":[],""urls"":[{""url"":""http://t.co/cJM5fzgoyL"",""expanded_url"":""http://vimeo.com/m/101936210"",""display_url"":""vimeo.com/m/101936210"",""indices"":[117,139]}],""user_mentions"":[{""screen_name"":""andrewxhill"",""name"":""Andrew W Hill"",""id"":19893224,""id_str"":""19893224"",""indices"":[103,115]}]}","note","object:search.twitter.com,2005:494274906528288769","Attention mappers: iframe, #javascript, app development examples from the #CartoDB API/team! Well done @andrewxhill. http://t.co/cJM5fzgoyL","2014-07-30T00:14:55.000Z","http://twitter.com/carygeo/statuses/494274906528288769","place","Los Angeles, CA","https://api.twitter.com/1.1/geo/id/3b77caf94bfc81fe.json","{""type"":""Polygon"",""coordinates"":[[[-118.668404,33.704538],[-118.668404,34.330724],[-118.155409,34.330724],[-118.155409,33.704538]]]}",,"Los Angeles","{""urls"":[{""url"":""http://t.co/cJM5fzgoyL"",""expanded_url"":""http://vimeo.com/101936210"",""expanded_status"":200}],""klout_score"":31,""language"":{""value"":""en""}}","{""type"":""Point"",""coordinates"":[-118.3620715,34.13846767]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:494194150133481472","post","http://twitter.com/juanjeojeda/statuses/494194150133481472","@dig_geo_com Your welcome :-) Keep in mind that some parts of @cartoDB still doesn't work fine. We're waiting for update from our provider.","activity","2014-07-29T18:54:01.000Z","0","medium","en","0","person","id:twitter.com:12679112","http://www.twitter.com/juanjeojeda","Juanje Ojeda ","https://pbs.twimg.com/profile_images/46025702/juanje_normal.jpg","Canarian guy who likes the open knowledge, the freedom and learn things every day. #OpenSource #DevOps #PaaS #Canarias","2008-01-25T11:49:15.000Z","[{""href"":""http://about.me/juanje"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Las Palmas de Gran Canaria""}","3600","juanjeojeda","[""es""]","London","416","1006","86","17700","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com","http://twitter.com/dig_geo_com/statuses/494191759078211584",,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""dig_geo_com"",""name"":""digital-geography"",""id"":231016902,""id_str"":""231016902"",""indices"":[0,12]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[62,70]}],""symbols"":[]}","note","object:search.twitter.com,2005:494194150133481472","@dig_geo_com Your welcome :-) Keep in mind that some parts of @cartoDB still doesn't work fine. We're waiting for update from our provider.","2014-07-29T18:54:01.000Z","http://twitter.com/juanjeojeda/statuses/494194150133481472","place","Las Palmas de Gran Canaria, Las Palmas","https://api.twitter.com/1.1/geo/id/00ab08eef1b62e92.json","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}",,"Las Palmas de Gran Canaria","{""klout_score"":56,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:494191583240388608","post","http://twitter.com/juanjeojeda/statuses/494191583240388608","@dig_geo_com Hi, @cartoDB blog is already back. We had some issues with our hosting provider. Sorry for the inconveniences.","activity","2014-07-29T18:43:49.000Z","0","medium","en","0","person","id:twitter.com:12679112","http://www.twitter.com/juanjeojeda","Juanje Ojeda ","https://pbs.twimg.com/profile_images/46025702/juanje_normal.jpg","Canarian guy who likes the open knowledge, the freedom and learn things every day. #OpenSource #DevOps #PaaS #Canarias","2008-01-25T11:49:15.000Z","[{""href"":""http://about.me/juanje"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Las Palmas de Gran Canaria""}","3600","juanjeojeda","[""es""]","London","416","1006","86","17698","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com","http://twitter.com/dig_geo_com/statuses/494181592685117442",,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""dig_geo_com"",""name"":""digital-geography"",""id"":231016902,""id_str"":""231016902"",""indices"":[0,12]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[17,25]}],""symbols"":[]}","note","object:search.twitter.com,2005:494191583240388608","@dig_geo_com Hi, @cartoDB blog is already back. We had some issues with our hosting provider. Sorry for the inconveniences.","2014-07-29T18:43:49.000Z","http://twitter.com/juanjeojeda/statuses/494191583240388608","place","Las Palmas de Gran Canaria, Las Palmas","https://api.twitter.com/1.1/geo/id/00ab08eef1b62e92.json","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}",,"Las Palmas de Gran Canaria","{""klout_score"":56,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:494141648922611712","post","http://twitter.com/spara/statuses/494141648922611712","@vtcraghead @briantimoney @billdollins CartoDB needs to cut a deal with a cloud provider, they’re a bit too parsimonious on disk space","activity","2014-07-29T15:25:24.000Z","0","medium","en","0","person","id:twitter.com:14629939","http://www.twitter.com/spara","spara","https://pbs.twimg.com/profile_images/469468370413166592/uuOHhLby_normal.jpeg","Ut mitterent eos in faciem.","2008-05-02T19:24:50.000Z","[{""href"":""http://sproke.blogspot.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""San Antonio Francisco""}","-25200","spara","[""en""]","Pacific Time (US & Canada)","1311","1276","128","24585","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com","http://twitter.com/vtcraghead/statuses/494140334226833408",,"{""hashtags"":[],""symbols"":[],""urls"":[],""user_mentions"":[{""screen_name"":""vtcraghead"",""name"":""Bill Morris"",""id"":278873782,""id_str"":""278873782"",""indices"":[0,11]},{""screen_name"":""briantimoney"",""name"":""Brian Timoney"",""id"":17546328,""id_str"":""17546328"",""indices"":[12,25]},{""screen_name"":""billdollins"",""name"":""Bill Dollins"",""id"":12405802,""id_str"":""12405802"",""indices"":[26,38]}]}","note","object:search.twitter.com,2005:494141648922611712","@vtcraghead @briantimoney @billdollins CartoDB needs to cut a deal with a cloud provider, they’re a bit too parsimonious on disk space","2014-07-29T15:25:24.000Z","http://twitter.com/spara/statuses/494141648922611712","place","Highland Park, San Antonio","https://api.twitter.com/1.1/geo/id/097c1754d9aa7b39.json","{""type"":""Polygon"",""coordinates"":[[[-98.478227,29.384112],[-98.478227,29.401578],[-98.448015,29.401578],[-98.448015,29.384112]]]}",,"Highland Park","{""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-98.478227,29.384112],[-98.478227,29.401578],[-98.448015,29.401578],[-98.448015,29.384112]]]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:496274441584668672","post","http://twitter.com/erictheise/statuses/496274441584668672","sad to be the guy that knocked ""locahost"" out of @cartoDB's docs for #Odyssey.js; i've certainly had that sentiment towards my dev machine.","activity","2014-08-04T12:40:21.000Z","0","medium","en","0","person","id:twitter.com:94817737","http://www.twitter.com/erictheise","Eric Theise","https://pbs.twimg.com/profile_images/1994194116/Theise_normal.jpg","Software engineer; photographer, experimental filmmaker, cartographer, geographer, vocalizer, yoga student, reader & writer, eater & drinker.","2009-12-05T16:06:12.000Z","[{""href"":null,""rel"":""me""}]","{""objectType"":""place"",""displayName"":""San Francisco""}","-25200","erictheise","[""en""]","Pacific Time (US & Canada)","698","306","25","1322","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[{""text"":""Odyssey"",""indices"":[69,77]}],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[49,57]}],""symbols"":[]}","note","object:search.twitter.com,2005:496274441584668672","sad to be the guy that knocked ""locahost"" out of @cartoDB's docs for #Odyssey.js; i've certainly had that sentiment towards my dev machine.","2014-08-04T12:40:21.000Z","http://twitter.com/erictheise/statuses/496274441584668672","place","San Francisco, CA","https://api.twitter.com/1.1/geo/id/5a110d312052166f.json","{""type"":""Polygon"",""coordinates"":[[[-122.514926,37.708075],[-122.514926,37.833238],[-122.328001,37.833238],[-122.328001,37.708075]]]}",,"San Francisco","{""klout_score"":41,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-122.514926,37.708075],[-122.514926,37.833238],[-122.328001,37.833238],[-122.328001,37.708075]]]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:495565956908064769","post","http://twitter.com/Jmholleran/statuses/495565956908064769","FrackMap #map http://t.co/BLWl3zYKTn","activity","2014-08-02T13:45:05.000Z","0","medium","en","0","person","id:twitter.com:27128139","http://www.twitter.com/Jmholleran","John Munro Holleran","https://pbs.twimg.com/profile_images/460662090819051520/lzLtBzUb_normal.png","Modest! intelligent, articulate & generally just amazing! Trade Unionist, learning facilitator 'n that","2009-03-27T23:36:39.000Z","[{""href"":null,""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Glasgow/Scotland""}",,"Jmholleran","[""en""]",,"411","192","1","1643","false","Twitter for Android","http://twitter.com/download/android","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[55.8641,-4.282299]}","{""hashtags"":[{""text"":""map"",""indices"":[9,13]}],""trends"":[],""urls"":[{""url"":""http://t.co/BLWl3zYKTn"",""expanded_url"":""http://energydesk.cartodb.com/viz/18162194-ecbd-11e3-aba1-0e230854a1cb/embed_map?title=true&description=true&search=false&shareable=true&cartodb_logo=true&layer_selector=false&legends=true&scrollwheel=true&fullscreen=true&sublayer_options=1%7C1%7C1%7C1&sql=&sw_lat=49.97948776108648&sw_lon=-12.7001953125&ne_lat=56.80087831233043&ne_lon=8.96484375"",""display_url"":""energydesk.cartodb.com/viz/18162194-e…"",""indices"":[14,36]}],""user_mentions"":[],""symbols"":[]}","note","object:search.twitter.com,2005:495565956908064769","FrackMap #map http://t.co/BLWl3zYKTn","2014-08-02T13:45:05.000Z","http://twitter.com/Jmholleran/statuses/495565956908064769","place","Glasgow","https://api.twitter.com/1.1/geo/id/791e00bcadc4615f.json","{""type"":""Polygon"",""coordinates"":[[[-4.3932845,55.796184],[-4.3932845,55.9204214],[-4.0902182,55.9204214],[-4.0902182,55.796184]]]}",,"Glasgow","{""urls"":[{""url"":""http://t.co/BLWl3zYKTn"",""expanded_url"":""http://energydesk.cartodb.com/viz/18162194-ecbd-11e3-aba1-0e230854a1cb/embed_map?title=true&description=true&search=false&shareable=true&cartodb_logo=true&layer_selector=false&legends=true&scrollwheel=true&fullscreen=true&sublayer_options=1%7C1%7C1%7C1&sql&sw_lat=49.97948776108648&sw_lon=-12.7001953125&ne_lat=56.80087831233043&ne_lon=8.96484375"",""expanded_status"":200}],""klout_score"":23}","{""type"":""Point"",""coordinates"":[-4.282299,55.8641]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:494521462682685440","post","http://twitter.com/SonOfJorEl/statuses/494521462682685440","@iriberri1 @cartoDB you are saying it is not supported, but you have an entire blog post about said feature : http://t.co/gvuUWv9sge","activity","2014-07-30T16:34:39.000Z","0","medium","en","0","person","id:twitter.com:15628310","http://www.twitter.com/SonOfJorEl","Joey Schluchter","https://pbs.twimg.com/profile_images/466697750013562880/ngwKpO1j_normal.jpeg","Recovering golfer. Helping feed the world to software. Still love persimmon.","2008-07-28T05:34:03.000Z","[{""href"":""http://www.exzeo.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Clearwater, FL""}","-14400","SonOfJorEl","[""en""]","Eastern Time (US & Canada)","368","2053","5","2143","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com","http://twitter.com/iriberri1/statuses/494488277765091328",,"{""hashtags"":[],""symbols"":[],""urls"":[{""url"":""http://t.co/gvuUWv9sge"",""expanded_url"":""http://blog.cartodb.com/got-files-weve-got-a-import-api/"",""display_url"":""blog.cartodb.com/got-files-weve…"",""indices"":[110,132]}],""user_mentions"":[{""screen_name"":""iriberri1"",""name"":""Carla"",""id"":102197411,""id_str"":""102197411"",""indices"":[0,10]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[11,19]}]}","note","object:search.twitter.com,2005:494521462682685440","@iriberri1 @cartoDB you are saying it is not supported, but you have an entire blog post about said feature : http://t.co/gvuUWv9sge","2014-07-30T16:34:39.000Z","http://twitter.com/SonOfJorEl/statuses/494521462682685440","place","Tampa, FL","https://api.twitter.com/1.1/geo/id/dc62519fda13b4ec.json","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}",,"Tampa","{""urls"":[{""url"":""http://t.co/gvuUWv9sge"",""expanded_url"":""http://blog.cartodb.com/got-files-weve-got-a-import-api/"",""expanded_status"":200}],""klout_score"":42,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:494511938563342336","post","http://twitter.com/xavijam/statuses/494511938563342336",". @Javier is giving birth last @cartodb feature. All I can say is … GIF over maps! http://t.co/oC3DEK5cP6","activity","2014-07-30T15:56:48.000Z","0","medium","en","0","person","id:twitter.com:26775253","http://www.twitter.com/xavijam","Javier Álvarez","https://pbs.twimg.com/profile_images/3025038956/a2919d353eb2b1a22756d7ac79847480_normal.jpeg","I love gentoo penguins","2009-03-26T15:36:52.000Z","[{""href"":""http://xavij.am"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Antarctic Peninsula""}","7200","xavijam","[""en""]","Madrid","250","395","26","4981","false","Tweetbot for iΟS","http://tapbots.com/tweetbot","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""javier"",""name"":""Javier Arce"",""id"":39083,""id_str"":""39083"",""indices"":[2,9]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[31,39]}],""symbols"":[],""media"":[{""id"":494511934629093378,""id_str"":""494511934629093378"",""indices"":[83,105],""media_url"":""http://pbs.twimg.com/media/Btzb_AjCIAI0cLr.jpg"",""media_url_https"":""https://pbs.twimg.com/media/Btzb_AjCIAI0cLr.jpg"",""url"":""http://t.co/oC3DEK5cP6"",""display_url"":""pic.twitter.com/oC3DEK5cP6"",""expanded_url"":""http://twitter.com/xavijam/status/494511938563342336/photo/1"",""type"":""photo"",""sizes"":{""small"":{""w"":340,""h"":255,""resize"":""fit""},""medium"":{""w"":600,""h"":450,""resize"":""fit""},""large"":{""w"":1024,""h"":768,""resize"":""fit""},""thumb"":{""w"":150,""h"":150,""resize"":""crop""}}}]}","note","object:search.twitter.com,2005:494511938563342336",". @Javier is giving birth last @cartodb feature. All I can say is … GIF over maps! http://t.co/oC3DEK5cP6","2014-07-30T15:56:48.000Z","http://twitter.com/xavijam/statuses/494511938563342336","place","Trafalgar","https://api.twitter.com/1.1/geo/id/0144b1172069289c.json","{""type"":""Polygon"",""coordinates"":[[[-3.70582,40.4275708],[-3.70582,40.4386747],[-3.6956326,40.4386747],[-3.6956326,40.4275708]]]}",,"Trafalgar","{""urls"":[{""url"":""http://t.co/oC3DEK5cP6"",""expanded_url"":""http://twitter.com/xavijam/status/494511938563342336/photo/1"",""expanded_status"":200}],""klout_score"":39,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-3.70582,40.4275708],[-3.70582,40.4386747],[-3.6956326,40.4386747],[-3.6956326,40.4275708]]]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:494485754848894978","post","http://twitter.com/SonOfJorEl/statuses/494485754848894978","@cartoDB I’ve emailed support (which you claim has a 24hr turnaround) with minimal response. I am a paying customer ($149 mo). For how long?","activity","2014-07-30T14:12:45.000Z","0","medium","en","0","person","id:twitter.com:15628310","http://www.twitter.com/SonOfJorEl","Joey Schluchter","https://pbs.twimg.com/profile_images/466697750013562880/ngwKpO1j_normal.jpeg","Recovering golfer. Helping feed the world to software. Still love persimmon.","2008-07-28T05:34:03.000Z","[{""href"":""http://www.exzeo.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Clearwater, FL""}","-14400","SonOfJorEl","[""en""]","Eastern Time (US & Canada)","368","2053","5","2142","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[0,8]}],""symbols"":[]}","note","object:search.twitter.com,2005:494485754848894978","@cartoDB I’ve emailed support (which you claim has a 24hr turnaround) with minimal response. I am a paying customer ($149 mo). For how long?","2014-07-30T14:12:45.000Z","http://twitter.com/SonOfJorEl/statuses/494485754848894978","place","Tampa, FL","https://api.twitter.com/1.1/geo/id/dc62519fda13b4ec.json","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}",,"Tampa","{""klout_score"":42,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:494276921765543936","post","http://twitter.com/httsan/statuses/494276921765543936","""@gisuser: Online Mapping for Beginners – The Map Academy from CartoDB: TweetSomething innovative from… http://t.co/Zhh882TV2F""","activity","2014-07-30T00:22:56.000Z","0","medium","en","0","person","id:twitter.com:28471178","http://www.twitter.com/httsan","hartanto","https://pbs.twimg.com/profile_images/2217327231/IMG0033_normal.jpg","Lahir di Indonesia | Nguli di @NEOnetBPPT @BPPTeknologi | Ngobrol di @RSGISForum @PadangZaitun | Nyantri di @MichiganStateU","2009-04-03T01:32:35.000Z","[{""href"":""http://about.me/httsan"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""East Lansing""}","-14400","httsan","[""en""]","Eastern Time (US & Canada)","976","873","12","25725","false","Twitter for Android","http://twitter.com/download/android","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[42.7539688,-84.4221123]}","{""hashtags"":[],""symbols"":[],""urls"":[{""url"":""http://t.co/Zhh882TV2F"",""expanded_url"":""http://goo.gl/fb/xVB5rr"",""display_url"":""goo.gl/fb/xVB5rr"",""indices"":[104,126]}],""user_mentions"":[{""screen_name"":""gisuser"",""name"":""GISuser GeoTech News"",""id"":16958615,""id_str"":""16958615"",""indices"":[1,9]}]}","note","object:search.twitter.com,2005:494276921765543936","""@gisuser: Online Mapping for Beginners – The Map Academy from CartoDB: TweetSomething innovative from… http://t.co/Zhh882TV2F""","2014-07-30T00:22:56.000Z","http://twitter.com/httsan/statuses/494276921765543936","place","Haslett, MI","https://api.twitter.com/1.1/geo/id/4399b5004a3b4d9a.json","{""type"":""Polygon"",""coordinates"":[[[-84.447506,42.731229],[-84.447506,42.769688],[-84.363432,42.769688],[-84.363432,42.731229]]]}",,"Haslett","{""urls"":[{""url"":""http://t.co/Zhh882TV2F"",""expanded_url"":""http://blog.gisuser.com/2014/07/29/online-mapping-for-beginners-the-map-academy-from-cartodb/?utm_source=feedburner&utm_medium=twitter&utm_campaign=Feed:%20gisuser%20(GISUser.com%20-%20GIS,%20Mapping,%20Geospatial,%20and%20location%20technology%20news)"",""expanded_status"":200}],""klout_score"":51,""language"":{""value"":""en""}}","{""type"":""Point"",""coordinates"":[-84.4221123,42.7539688]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:494274906528288769","post","http://twitter.com/carygeo/statuses/494274906528288769","Attention mappers: iframe, #javascript, app development examples from the #CartoDB API/team! Well done @andrewxhill. http://t.co/cJM5fzgoyL","activity","2014-07-30T00:14:55.000Z","0","medium","en","0","person","id:twitter.com:2641688154","http://www.twitter.com/carygeo","Cary Greenwood","https://pbs.twimg.com/profile_images/488501325697544192/gFMxHuLV_normal.jpeg","geography, maps, data visualization, ui/ux design","2014-07-14T01:35:22.000Z","[{""href"":null,""rel"":""me""}]",,,"carygeo","[""en""]",,"109","46","0","47","false","Twitter for iPhone","http://twitter.com/download/iphone","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[34.13846767,-118.3620715]}","{""hashtags"":[{""text"":""javascript"",""indices"":[27,38]},{""text"":""CartoDB"",""indices"":[74,82]}],""symbols"":[],""urls"":[{""url"":""http://t.co/cJM5fzgoyL"",""expanded_url"":""http://vimeo.com/m/101936210"",""display_url"":""vimeo.com/m/101936210"",""indices"":[117,139]}],""user_mentions"":[{""screen_name"":""andrewxhill"",""name"":""Andrew W Hill"",""id"":19893224,""id_str"":""19893224"",""indices"":[103,115]}]}","note","object:search.twitter.com,2005:494274906528288769","Attention mappers: iframe, #javascript, app development examples from the #CartoDB API/team! Well done @andrewxhill. http://t.co/cJM5fzgoyL","2014-07-30T00:14:55.000Z","http://twitter.com/carygeo/statuses/494274906528288769","place","Los Angeles, CA","https://api.twitter.com/1.1/geo/id/3b77caf94bfc81fe.json","{""type"":""Polygon"",""coordinates"":[[[-118.668404,33.704538],[-118.668404,34.330724],[-118.155409,34.330724],[-118.155409,33.704538]]]}",,"Los Angeles","{""urls"":[{""url"":""http://t.co/cJM5fzgoyL"",""expanded_url"":""http://vimeo.com/101936210"",""expanded_status"":200}],""klout_score"":31,""language"":{""value"":""en""}}","{""type"":""Point"",""coordinates"":[-118.3620715,34.13846767]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:494194150133481472","post","http://twitter.com/juanjeojeda/statuses/494194150133481472","@dig_geo_com Your welcome :-) Keep in mind that some parts of @cartoDB still doesn't work fine. We're waiting for update from our provider.","activity","2014-07-29T18:54:01.000Z","0","medium","en","0","person","id:twitter.com:12679112","http://www.twitter.com/juanjeojeda","Juanje Ojeda ","https://pbs.twimg.com/profile_images/46025702/juanje_normal.jpg","Canarian guy who likes the open knowledge, the freedom and learn things every day. #OpenSource #DevOps #PaaS #Canarias","2008-01-25T11:49:15.000Z","[{""href"":""http://about.me/juanje"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Las Palmas de Gran Canaria""}","3600","juanjeojeda","[""es""]","London","416","1006","86","17700","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com","http://twitter.com/dig_geo_com/statuses/494191759078211584",,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""dig_geo_com"",""name"":""digital-geography"",""id"":231016902,""id_str"":""231016902"",""indices"":[0,12]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[62,70]}],""symbols"":[]}","note","object:search.twitter.com,2005:494194150133481472","@dig_geo_com Your welcome :-) Keep in mind that some parts of @cartoDB still doesn't work fine. We're waiting for update from our provider.","2014-07-29T18:54:01.000Z","http://twitter.com/juanjeojeda/statuses/494194150133481472","place","Las Palmas de Gran Canaria, Las Palmas","https://api.twitter.com/1.1/geo/id/00ab08eef1b62e92.json","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}",,"Las Palmas de Gran Canaria","{""klout_score"":56,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:494191583240388608","post","http://twitter.com/juanjeojeda/statuses/494191583240388608","@dig_geo_com Hi, @cartoDB blog is already back. We had some issues with our hosting provider. Sorry for the inconveniences.","activity","2014-07-29T18:43:49.000Z","0","medium","en","0","person","id:twitter.com:12679112","http://www.twitter.com/juanjeojeda","Juanje Ojeda ","https://pbs.twimg.com/profile_images/46025702/juanje_normal.jpg","Canarian guy who likes the open knowledge, the freedom and learn things every day. #OpenSource #DevOps #PaaS #Canarias","2008-01-25T11:49:15.000Z","[{""href"":""http://about.me/juanje"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Las Palmas de Gran Canaria""}","3600","juanjeojeda","[""es""]","London","416","1006","86","17698","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com","http://twitter.com/dig_geo_com/statuses/494181592685117442",,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""dig_geo_com"",""name"":""digital-geography"",""id"":231016902,""id_str"":""231016902"",""indices"":[0,12]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[17,25]}],""symbols"":[]}","note","object:search.twitter.com,2005:494191583240388608","@dig_geo_com Hi, @cartoDB blog is already back. We had some issues with our hosting provider. Sorry for the inconveniences.","2014-07-29T18:43:49.000Z","http://twitter.com/juanjeojeda/statuses/494191583240388608","place","Las Palmas de Gran Canaria, Las Palmas","https://api.twitter.com/1.1/geo/id/00ab08eef1b62e92.json","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}",,"Las Palmas de Gran Canaria","{""klout_score"":56,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:494141648922611712","post","http://twitter.com/spara/statuses/494141648922611712","@vtcraghead @briantimoney @billdollins CartoDB needs to cut a deal with a cloud provider, they’re a bit too parsimonious on disk space","activity","2014-07-29T15:25:24.000Z","0","medium","en","0","person","id:twitter.com:14629939","http://www.twitter.com/spara","spara","https://pbs.twimg.com/profile_images/469468370413166592/uuOHhLby_normal.jpeg","Ut mitterent eos in faciem.","2008-05-02T19:24:50.000Z","[{""href"":""http://sproke.blogspot.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""San Antonio Francisco""}","-25200","spara","[""en""]","Pacific Time (US & Canada)","1311","1276","128","24585","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com","http://twitter.com/vtcraghead/statuses/494140334226833408",,"{""hashtags"":[],""symbols"":[],""urls"":[],""user_mentions"":[{""screen_name"":""vtcraghead"",""name"":""Bill Morris"",""id"":278873782,""id_str"":""278873782"",""indices"":[0,11]},{""screen_name"":""briantimoney"",""name"":""Brian Timoney"",""id"":17546328,""id_str"":""17546328"",""indices"":[12,25]},{""screen_name"":""billdollins"",""name"":""Bill Dollins"",""id"":12405802,""id_str"":""12405802"",""indices"":[26,38]}]}","note","object:search.twitter.com,2005:494141648922611712","@vtcraghead @briantimoney @billdollins CartoDB needs to cut a deal with a cloud provider, they’re a bit too parsimonious on disk space","2014-07-29T15:25:24.000Z","http://twitter.com/spara/statuses/494141648922611712","place","Highland Park, San Antonio","https://api.twitter.com/1.1/geo/id/097c1754d9aa7b39.json","{""type"":""Polygon"",""coordinates"":[[[-98.478227,29.384112],[-98.478227,29.401578],[-98.448015,29.401578],[-98.448015,29.384112]]]}",,"Highland Park","{""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-98.478227,29.384112],[-98.478227,29.401578],[-98.448015,29.401578],[-98.448015,29.384112]]]}","Category 1","uno, @dos, #tres"
|
||||
"tag:search.twitter.com,2005:496274441584668672","post","http://twitter.com/erictheise/statuses/496274441584668672","sad to be the guy that knocked ""locahost"" out of @cartoDB's docs for #Odyssey.js; i've certainly had that sentiment towards my dev machine.","activity","2014-08-04T12:40:21.000Z","0","medium","en","0","person","id:twitter.com:94817737","http://www.twitter.com/erictheise","Eric Theise","https://pbs.twimg.com/profile_images/1994194116/Theise_normal.jpg","Software engineer; photographer, experimental filmmaker, cartographer, geographer, vocalizer, yoga student, reader & writer, eater & drinker.","2009-12-05T16:06:12.000Z","[{""href"":null,""rel"":""me""}]","{""objectType"":""place"",""displayName"":""San Francisco""}","-25200","erictheise","[""en""]","Pacific Time (US & Canada)","698","306","25","1322","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[{""text"":""Odyssey"",""indices"":[69,77]}],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[49,57]}],""symbols"":[]}","note","object:search.twitter.com,2005:496274441584668672","sad to be the guy that knocked ""locahost"" out of @cartoDB's docs for #Odyssey.js; i've certainly had that sentiment towards my dev machine.","2014-08-04T12:40:21.000Z","http://twitter.com/erictheise/statuses/496274441584668672","place","San Francisco, CA","https://api.twitter.com/1.1/geo/id/5a110d312052166f.json","{""type"":""Polygon"",""coordinates"":[[[-122.514926,37.708075],[-122.514926,37.833238],[-122.328001,37.833238],[-122.328001,37.708075]]]}",,"San Francisco","{""klout_score"":41,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-122.514926,37.708075],[-122.514926,37.833238],[-122.328001,37.833238],[-122.328001,37.708075]]]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:495565956908064769","post","http://twitter.com/Jmholleran/statuses/495565956908064769","FrackMap #map http://t.co/BLWl3zYKTn","activity","2014-08-02T13:45:05.000Z","0","medium","en","0","person","id:twitter.com:27128139","http://www.twitter.com/Jmholleran","John Munro Holleran","https://pbs.twimg.com/profile_images/460662090819051520/lzLtBzUb_normal.png","Modest! intelligent, articulate & generally just amazing! Trade Unionist, learning facilitator 'n that","2009-03-27T23:36:39.000Z","[{""href"":null,""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Glasgow/Scotland""}",,"Jmholleran","[""en""]",,"411","192","1","1643","false","Twitter for Android","http://twitter.com/download/android","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[55.8641,-4.282299]}","{""hashtags"":[{""text"":""map"",""indices"":[9,13]}],""trends"":[],""urls"":[{""url"":""http://t.co/BLWl3zYKTn"",""expanded_url"":""http://energydesk.cartodb.com/viz/18162194-ecbd-11e3-aba1-0e230854a1cb/embed_map?title=true&description=true&search=false&shareable=true&cartodb_logo=true&layer_selector=false&legends=true&scrollwheel=true&fullscreen=true&sublayer_options=1%7C1%7C1%7C1&sql=&sw_lat=49.97948776108648&sw_lon=-12.7001953125&ne_lat=56.80087831233043&ne_lon=8.96484375"",""display_url"":""energydesk.cartodb.com/viz/18162194-e…"",""indices"":[14,36]}],""user_mentions"":[],""symbols"":[]}","note","object:search.twitter.com,2005:495565956908064769","FrackMap #map http://t.co/BLWl3zYKTn","2014-08-02T13:45:05.000Z","http://twitter.com/Jmholleran/statuses/495565956908064769","place","Glasgow","https://api.twitter.com/1.1/geo/id/791e00bcadc4615f.json","{""type"":""Polygon"",""coordinates"":[[[-4.3932845,55.796184],[-4.3932845,55.9204214],[-4.0902182,55.9204214],[-4.0902182,55.796184]]]}",,"Glasgow","{""urls"":[{""url"":""http://t.co/BLWl3zYKTn"",""expanded_url"":""http://energydesk.cartodb.com/viz/18162194-ecbd-11e3-aba1-0e230854a1cb/embed_map?title=true&description=true&search=false&shareable=true&cartodb_logo=true&layer_selector=false&legends=true&scrollwheel=true&fullscreen=true&sublayer_options=1%7C1%7C1%7C1&sql&sw_lat=49.97948776108648&sw_lon=-12.7001953125&ne_lat=56.80087831233043&ne_lon=8.96484375"",""expanded_status"":200}],""klout_score"":23}","{""type"":""Point"",""coordinates"":[-4.282299,55.8641]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:494521462682685440","post","http://twitter.com/SonOfJorEl/statuses/494521462682685440","@iriberri1 @cartoDB you are saying it is not supported, but you have an entire blog post about said feature : http://t.co/gvuUWv9sge","activity","2014-07-30T16:34:39.000Z","0","medium","en","0","person","id:twitter.com:15628310","http://www.twitter.com/SonOfJorEl","Joey Schluchter","https://pbs.twimg.com/profile_images/466697750013562880/ngwKpO1j_normal.jpeg","Recovering golfer. Helping feed the world to software. Still love persimmon.","2008-07-28T05:34:03.000Z","[{""href"":""http://www.exzeo.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Clearwater, FL""}","-14400","SonOfJorEl","[""en""]","Eastern Time (US & Canada)","368","2053","5","2143","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com","http://twitter.com/iriberri1/statuses/494488277765091328",,"{""hashtags"":[],""symbols"":[],""urls"":[{""url"":""http://t.co/gvuUWv9sge"",""expanded_url"":""http://blog.cartodb.com/got-files-weve-got-a-import-api/"",""display_url"":""blog.cartodb.com/got-files-weve…"",""indices"":[110,132]}],""user_mentions"":[{""screen_name"":""iriberri1"",""name"":""Carla"",""id"":102197411,""id_str"":""102197411"",""indices"":[0,10]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[11,19]}]}","note","object:search.twitter.com,2005:494521462682685440","@iriberri1 @cartoDB you are saying it is not supported, but you have an entire blog post about said feature : http://t.co/gvuUWv9sge","2014-07-30T16:34:39.000Z","http://twitter.com/SonOfJorEl/statuses/494521462682685440","place","Tampa, FL","https://api.twitter.com/1.1/geo/id/dc62519fda13b4ec.json","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}",,"Tampa","{""urls"":[{""url"":""http://t.co/gvuUWv9sge"",""expanded_url"":""http://blog.cartodb.com/got-files-weve-got-a-import-api/"",""expanded_status"":200}],""klout_score"":42,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:494511938563342336","post","http://twitter.com/xavijam/statuses/494511938563342336",". @Javier is giving birth last @cartodb feature. All I can say is … GIF over maps! http://t.co/oC3DEK5cP6","activity","2014-07-30T15:56:48.000Z","0","medium","en","0","person","id:twitter.com:26775253","http://www.twitter.com/xavijam","Javier Álvarez","https://pbs.twimg.com/profile_images/3025038956/a2919d353eb2b1a22756d7ac79847480_normal.jpeg","I love gentoo penguins","2009-03-26T15:36:52.000Z","[{""href"":""http://xavij.am"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Antarctic Peninsula""}","7200","xavijam","[""en""]","Madrid","250","395","26","4981","false","Tweetbot for iΟS","http://tapbots.com/tweetbot","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""javier"",""name"":""Javier Arce"",""id"":39083,""id_str"":""39083"",""indices"":[2,9]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[31,39]}],""symbols"":[],""media"":[{""id"":494511934629093378,""id_str"":""494511934629093378"",""indices"":[83,105],""media_url"":""http://pbs.twimg.com/media/Btzb_AjCIAI0cLr.jpg"",""media_url_https"":""https://pbs.twimg.com/media/Btzb_AjCIAI0cLr.jpg"",""url"":""http://t.co/oC3DEK5cP6"",""display_url"":""pic.twitter.com/oC3DEK5cP6"",""expanded_url"":""http://twitter.com/xavijam/status/494511938563342336/photo/1"",""type"":""photo"",""sizes"":{""small"":{""w"":340,""h"":255,""resize"":""fit""},""medium"":{""w"":600,""h"":450,""resize"":""fit""},""large"":{""w"":1024,""h"":768,""resize"":""fit""},""thumb"":{""w"":150,""h"":150,""resize"":""crop""}}}]}","note","object:search.twitter.com,2005:494511938563342336",". @Javier is giving birth last @cartodb feature. All I can say is … GIF over maps! http://t.co/oC3DEK5cP6","2014-07-30T15:56:48.000Z","http://twitter.com/xavijam/statuses/494511938563342336","place","Trafalgar","https://api.twitter.com/1.1/geo/id/0144b1172069289c.json","{""type"":""Polygon"",""coordinates"":[[[-3.70582,40.4275708],[-3.70582,40.4386747],[-3.6956326,40.4386747],[-3.6956326,40.4275708]]]}",,"Trafalgar","{""urls"":[{""url"":""http://t.co/oC3DEK5cP6"",""expanded_url"":""http://twitter.com/xavijam/status/494511938563342336/photo/1"",""expanded_status"":200}],""klout_score"":39,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-3.70582,40.4275708],[-3.70582,40.4386747],[-3.6956326,40.4386747],[-3.6956326,40.4275708]]]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:494485754848894978","post","http://twitter.com/SonOfJorEl/statuses/494485754848894978","@cartoDB I’ve emailed support (which you claim has a 24hr turnaround) with minimal response. I am a paying customer ($149 mo). For how long?","activity","2014-07-30T14:12:45.000Z","0","medium","en","0","person","id:twitter.com:15628310","http://www.twitter.com/SonOfJorEl","Joey Schluchter","https://pbs.twimg.com/profile_images/466697750013562880/ngwKpO1j_normal.jpeg","Recovering golfer. Helping feed the world to software. Still love persimmon.","2008-07-28T05:34:03.000Z","[{""href"":""http://www.exzeo.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Clearwater, FL""}","-14400","SonOfJorEl","[""en""]","Eastern Time (US & Canada)","368","2053","5","2142","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[0,8]}],""symbols"":[]}","note","object:search.twitter.com,2005:494485754848894978","@cartoDB I’ve emailed support (which you claim has a 24hr turnaround) with minimal response. I am a paying customer ($149 mo). For how long?","2014-07-30T14:12:45.000Z","http://twitter.com/SonOfJorEl/statuses/494485754848894978","place","Tampa, FL","https://api.twitter.com/1.1/geo/id/dc62519fda13b4ec.json","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}",,"Tampa","{""klout_score"":42,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:494276921765543936","post","http://twitter.com/httsan/statuses/494276921765543936","""@gisuser: Online Mapping for Beginners – The Map Academy from CartoDB: TweetSomething innovative from… http://t.co/Zhh882TV2F""","activity","2014-07-30T00:22:56.000Z","0","medium","en","0","person","id:twitter.com:28471178","http://www.twitter.com/httsan","hartanto","https://pbs.twimg.com/profile_images/2217327231/IMG0033_normal.jpg","Lahir di Indonesia | Nguli di @NEOnetBPPT @BPPTeknologi | Ngobrol di @RSGISForum @PadangZaitun | Nyantri di @MichiganStateU","2009-04-03T01:32:35.000Z","[{""href"":""http://about.me/httsan"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""East Lansing""}","-14400","httsan","[""en""]","Eastern Time (US & Canada)","976","873","12","25725","false","Twitter for Android","http://twitter.com/download/android","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[42.7539688,-84.4221123]}","{""hashtags"":[],""symbols"":[],""urls"":[{""url"":""http://t.co/Zhh882TV2F"",""expanded_url"":""http://goo.gl/fb/xVB5rr"",""display_url"":""goo.gl/fb/xVB5rr"",""indices"":[104,126]}],""user_mentions"":[{""screen_name"":""gisuser"",""name"":""GISuser GeoTech News"",""id"":16958615,""id_str"":""16958615"",""indices"":[1,9]}]}","note","object:search.twitter.com,2005:494276921765543936","""@gisuser: Online Mapping for Beginners – The Map Academy from CartoDB: TweetSomething innovative from… http://t.co/Zhh882TV2F""","2014-07-30T00:22:56.000Z","http://twitter.com/httsan/statuses/494276921765543936","place","Haslett, MI","https://api.twitter.com/1.1/geo/id/4399b5004a3b4d9a.json","{""type"":""Polygon"",""coordinates"":[[[-84.447506,42.731229],[-84.447506,42.769688],[-84.363432,42.769688],[-84.363432,42.731229]]]}",,"Haslett","{""urls"":[{""url"":""http://t.co/Zhh882TV2F"",""expanded_url"":""http://blog.gisuser.com/2014/07/29/online-mapping-for-beginners-the-map-academy-from-cartodb/?utm_source=feedburner&utm_medium=twitter&utm_campaign=Feed:%20gisuser%20(GISUser.com%20-%20GIS,%20Mapping,%20Geospatial,%20and%20location%20technology%20news)"",""expanded_status"":200}],""klout_score"":51,""language"":{""value"":""en""}}","{""type"":""Point"",""coordinates"":[-84.4221123,42.7539688]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:494274906528288769","post","http://twitter.com/carygeo/statuses/494274906528288769","Attention mappers: iframe, #javascript, app development examples from the #CartoDB API/team! Well done @andrewxhill. http://t.co/cJM5fzgoyL","activity","2014-07-30T00:14:55.000Z","0","medium","en","0","person","id:twitter.com:2641688154","http://www.twitter.com/carygeo","Cary Greenwood","https://pbs.twimg.com/profile_images/488501325697544192/gFMxHuLV_normal.jpeg","geography, maps, data visualization, ui/ux design","2014-07-14T01:35:22.000Z","[{""href"":null,""rel"":""me""}]",,,"carygeo","[""en""]",,"109","46","0","47","false","Twitter for iPhone","http://twitter.com/download/iphone","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[34.13846767,-118.3620715]}","{""hashtags"":[{""text"":""javascript"",""indices"":[27,38]},{""text"":""CartoDB"",""indices"":[74,82]}],""symbols"":[],""urls"":[{""url"":""http://t.co/cJM5fzgoyL"",""expanded_url"":""http://vimeo.com/m/101936210"",""display_url"":""vimeo.com/m/101936210"",""indices"":[117,139]}],""user_mentions"":[{""screen_name"":""andrewxhill"",""name"":""Andrew W Hill"",""id"":19893224,""id_str"":""19893224"",""indices"":[103,115]}]}","note","object:search.twitter.com,2005:494274906528288769","Attention mappers: iframe, #javascript, app development examples from the #CartoDB API/team! Well done @andrewxhill. http://t.co/cJM5fzgoyL","2014-07-30T00:14:55.000Z","http://twitter.com/carygeo/statuses/494274906528288769","place","Los Angeles, CA","https://api.twitter.com/1.1/geo/id/3b77caf94bfc81fe.json","{""type"":""Polygon"",""coordinates"":[[[-118.668404,33.704538],[-118.668404,34.330724],[-118.155409,34.330724],[-118.155409,33.704538]]]}",,"Los Angeles","{""urls"":[{""url"":""http://t.co/cJM5fzgoyL"",""expanded_url"":""http://vimeo.com/101936210"",""expanded_status"":200}],""klout_score"":31,""language"":{""value"":""en""}}","{""type"":""Point"",""coordinates"":[-118.3620715,34.13846767]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:494194150133481472","post","http://twitter.com/juanjeojeda/statuses/494194150133481472","@dig_geo_com Your welcome :-) Keep in mind that some parts of @cartoDB still doesn't work fine. We're waiting for update from our provider.","activity","2014-07-29T18:54:01.000Z","0","medium","en","0","person","id:twitter.com:12679112","http://www.twitter.com/juanjeojeda","Juanje Ojeda ","https://pbs.twimg.com/profile_images/46025702/juanje_normal.jpg","Canarian guy who likes the open knowledge, the freedom and learn things every day. #OpenSource #DevOps #PaaS #Canarias","2008-01-25T11:49:15.000Z","[{""href"":""http://about.me/juanje"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Las Palmas de Gran Canaria""}","3600","juanjeojeda","[""es""]","London","416","1006","86","17700","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com","http://twitter.com/dig_geo_com/statuses/494191759078211584",,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""dig_geo_com"",""name"":""digital-geography"",""id"":231016902,""id_str"":""231016902"",""indices"":[0,12]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[62,70]}],""symbols"":[]}","note","object:search.twitter.com,2005:494194150133481472","@dig_geo_com Your welcome :-) Keep in mind that some parts of @cartoDB still doesn't work fine. We're waiting for update from our provider.","2014-07-29T18:54:01.000Z","http://twitter.com/juanjeojeda/statuses/494194150133481472","place","Las Palmas de Gran Canaria, Las Palmas","https://api.twitter.com/1.1/geo/id/00ab08eef1b62e92.json","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}",,"Las Palmas de Gran Canaria","{""klout_score"":56,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:494191583240388608","post","http://twitter.com/juanjeojeda/statuses/494191583240388608","@dig_geo_com Hi, @cartoDB blog is already back. We had some issues with our hosting provider. Sorry for the inconveniences.","activity","2014-07-29T18:43:49.000Z","0","medium","en","0","person","id:twitter.com:12679112","http://www.twitter.com/juanjeojeda","Juanje Ojeda ","https://pbs.twimg.com/profile_images/46025702/juanje_normal.jpg","Canarian guy who likes the open knowledge, the freedom and learn things every day. #OpenSource #DevOps #PaaS #Canarias","2008-01-25T11:49:15.000Z","[{""href"":""http://about.me/juanje"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Las Palmas de Gran Canaria""}","3600","juanjeojeda","[""es""]","London","416","1006","86","17698","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com","http://twitter.com/dig_geo_com/statuses/494181592685117442",,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""dig_geo_com"",""name"":""digital-geography"",""id"":231016902,""id_str"":""231016902"",""indices"":[0,12]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[17,25]}],""symbols"":[]}","note","object:search.twitter.com,2005:494191583240388608","@dig_geo_com Hi, @cartoDB blog is already back. We had some issues with our hosting provider. Sorry for the inconveniences.","2014-07-29T18:43:49.000Z","http://twitter.com/juanjeojeda/statuses/494191583240388608","place","Las Palmas de Gran Canaria, Las Palmas","https://api.twitter.com/1.1/geo/id/00ab08eef1b62e92.json","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}",,"Las Palmas de Gran Canaria","{""klout_score"":56,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:494141648922611712","post","http://twitter.com/spara/statuses/494141648922611712","@vtcraghead @briantimoney @billdollins CartoDB needs to cut a deal with a cloud provider, they’re a bit too parsimonious on disk space","activity","2014-07-29T15:25:24.000Z","0","medium","en","0","person","id:twitter.com:14629939","http://www.twitter.com/spara","spara","https://pbs.twimg.com/profile_images/469468370413166592/uuOHhLby_normal.jpeg","Ut mitterent eos in faciem.","2008-05-02T19:24:50.000Z","[{""href"":""http://sproke.blogspot.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""San Antonio Francisco""}","-25200","spara","[""en""]","Pacific Time (US & Canada)","1311","1276","128","24585","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com","http://twitter.com/vtcraghead/statuses/494140334226833408",,"{""hashtags"":[],""symbols"":[],""urls"":[],""user_mentions"":[{""screen_name"":""vtcraghead"",""name"":""Bill Morris"",""id"":278873782,""id_str"":""278873782"",""indices"":[0,11]},{""screen_name"":""briantimoney"",""name"":""Brian Timoney"",""id"":17546328,""id_str"":""17546328"",""indices"":[12,25]},{""screen_name"":""billdollins"",""name"":""Bill Dollins"",""id"":12405802,""id_str"":""12405802"",""indices"":[26,38]}]}","note","object:search.twitter.com,2005:494141648922611712","@vtcraghead @briantimoney @billdollins CartoDB needs to cut a deal with a cloud provider, they’re a bit too parsimonious on disk space","2014-07-29T15:25:24.000Z","http://twitter.com/spara/statuses/494141648922611712","place","Highland Park, San Antonio","https://api.twitter.com/1.1/geo/id/097c1754d9aa7b39.json","{""type"":""Polygon"",""coordinates"":[[[-98.478227,29.384112],[-98.478227,29.401578],[-98.448015,29.401578],[-98.448015,29.384112]]]}",,"Highland Park","{""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-98.478227,29.384112],[-98.478227,29.401578],[-98.448015,29.401578],[-98.448015,29.384112]]]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:496274441584668672","post","http://twitter.com/erictheise/statuses/496274441584668672","sad to be the guy that knocked ""locahost"" out of @cartoDB's docs for #Odyssey.js; i've certainly had that sentiment towards my dev machine.","activity","2014-08-04T12:40:21.000Z","0","medium","en","0","person","id:twitter.com:94817737","http://www.twitter.com/erictheise","Eric Theise","https://pbs.twimg.com/profile_images/1994194116/Theise_normal.jpg","Software engineer; photographer, experimental filmmaker, cartographer, geographer, vocalizer, yoga student, reader & writer, eater & drinker.","2009-12-05T16:06:12.000Z","[{""href"":null,""rel"":""me""}]","{""objectType"":""place"",""displayName"":""San Francisco""}","-25200","erictheise","[""en""]","Pacific Time (US & Canada)","698","306","25","1322","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[{""text"":""Odyssey"",""indices"":[69,77]}],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[49,57]}],""symbols"":[]}","note","object:search.twitter.com,2005:496274441584668672","sad to be the guy that knocked ""locahost"" out of @cartoDB's docs for #Odyssey.js; i've certainly had that sentiment towards my dev machine.","2014-08-04T12:40:21.000Z","http://twitter.com/erictheise/statuses/496274441584668672","place","San Francisco, CA","https://api.twitter.com/1.1/geo/id/5a110d312052166f.json","{""type"":""Polygon"",""coordinates"":[[[-122.514926,37.708075],[-122.514926,37.833238],[-122.328001,37.833238],[-122.328001,37.708075]]]}",,"San Francisco","{""klout_score"":41,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-122.514926,37.708075],[-122.514926,37.833238],[-122.328001,37.833238],[-122.328001,37.708075]]]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:495565956908064769","post","http://twitter.com/Jmholleran/statuses/495565956908064769","FrackMap #map http://t.co/BLWl3zYKTn","activity","2014-08-02T13:45:05.000Z","0","medium","en","0","person","id:twitter.com:27128139","http://www.twitter.com/Jmholleran","John Munro Holleran","https://pbs.twimg.com/profile_images/460662090819051520/lzLtBzUb_normal.png","Modest! intelligent, articulate & generally just amazing! Trade Unionist, learning facilitator 'n that","2009-03-27T23:36:39.000Z","[{""href"":null,""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Glasgow/Scotland""}",,"Jmholleran","[""en""]",,"411","192","1","1643","false","Twitter for Android","http://twitter.com/download/android","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[55.8641,-4.282299]}","{""hashtags"":[{""text"":""map"",""indices"":[9,13]}],""trends"":[],""urls"":[{""url"":""http://t.co/BLWl3zYKTn"",""expanded_url"":""http://energydesk.cartodb.com/viz/18162194-ecbd-11e3-aba1-0e230854a1cb/embed_map?title=true&description=true&search=false&shareable=true&cartodb_logo=true&layer_selector=false&legends=true&scrollwheel=true&fullscreen=true&sublayer_options=1%7C1%7C1%7C1&sql=&sw_lat=49.97948776108648&sw_lon=-12.7001953125&ne_lat=56.80087831233043&ne_lon=8.96484375"",""display_url"":""energydesk.cartodb.com/viz/18162194-e…"",""indices"":[14,36]}],""user_mentions"":[],""symbols"":[]}","note","object:search.twitter.com,2005:495565956908064769","FrackMap #map http://t.co/BLWl3zYKTn","2014-08-02T13:45:05.000Z","http://twitter.com/Jmholleran/statuses/495565956908064769","place","Glasgow","https://api.twitter.com/1.1/geo/id/791e00bcadc4615f.json","{""type"":""Polygon"",""coordinates"":[[[-4.3932845,55.796184],[-4.3932845,55.9204214],[-4.0902182,55.9204214],[-4.0902182,55.796184]]]}",,"Glasgow","{""urls"":[{""url"":""http://t.co/BLWl3zYKTn"",""expanded_url"":""http://energydesk.cartodb.com/viz/18162194-ecbd-11e3-aba1-0e230854a1cb/embed_map?title=true&description=true&search=false&shareable=true&cartodb_logo=true&layer_selector=false&legends=true&scrollwheel=true&fullscreen=true&sublayer_options=1%7C1%7C1%7C1&sql&sw_lat=49.97948776108648&sw_lon=-12.7001953125&ne_lat=56.80087831233043&ne_lon=8.96484375"",""expanded_status"":200}],""klout_score"":23}","{""type"":""Point"",""coordinates"":[-4.282299,55.8641]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:494521462682685440","post","http://twitter.com/SonOfJorEl/statuses/494521462682685440","@iriberri1 @cartoDB you are saying it is not supported, but you have an entire blog post about said feature : http://t.co/gvuUWv9sge","activity","2014-07-30T16:34:39.000Z","0","medium","en","0","person","id:twitter.com:15628310","http://www.twitter.com/SonOfJorEl","Joey Schluchter","https://pbs.twimg.com/profile_images/466697750013562880/ngwKpO1j_normal.jpeg","Recovering golfer. Helping feed the world to software. Still love persimmon.","2008-07-28T05:34:03.000Z","[{""href"":""http://www.exzeo.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Clearwater, FL""}","-14400","SonOfJorEl","[""en""]","Eastern Time (US & Canada)","368","2053","5","2143","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com","http://twitter.com/iriberri1/statuses/494488277765091328",,"{""hashtags"":[],""symbols"":[],""urls"":[{""url"":""http://t.co/gvuUWv9sge"",""expanded_url"":""http://blog.cartodb.com/got-files-weve-got-a-import-api/"",""display_url"":""blog.cartodb.com/got-files-weve…"",""indices"":[110,132]}],""user_mentions"":[{""screen_name"":""iriberri1"",""name"":""Carla"",""id"":102197411,""id_str"":""102197411"",""indices"":[0,10]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[11,19]}]}","note","object:search.twitter.com,2005:494521462682685440","@iriberri1 @cartoDB you are saying it is not supported, but you have an entire blog post about said feature : http://t.co/gvuUWv9sge","2014-07-30T16:34:39.000Z","http://twitter.com/SonOfJorEl/statuses/494521462682685440","place","Tampa, FL","https://api.twitter.com/1.1/geo/id/dc62519fda13b4ec.json","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}",,"Tampa","{""urls"":[{""url"":""http://t.co/gvuUWv9sge"",""expanded_url"":""http://blog.cartodb.com/got-files-weve-got-a-import-api/"",""expanded_status"":200}],""klout_score"":42,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:494511938563342336","post","http://twitter.com/xavijam/statuses/494511938563342336",". @Javier is giving birth last @cartodb feature. All I can say is … GIF over maps! http://t.co/oC3DEK5cP6","activity","2014-07-30T15:56:48.000Z","0","medium","en","0","person","id:twitter.com:26775253","http://www.twitter.com/xavijam","Javier Álvarez","https://pbs.twimg.com/profile_images/3025038956/a2919d353eb2b1a22756d7ac79847480_normal.jpeg","I love gentoo penguins","2009-03-26T15:36:52.000Z","[{""href"":""http://xavij.am"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Antarctic Peninsula""}","7200","xavijam","[""en""]","Madrid","250","395","26","4981","false","Tweetbot for iΟS","http://tapbots.com/tweetbot","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""javier"",""name"":""Javier Arce"",""id"":39083,""id_str"":""39083"",""indices"":[2,9]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[31,39]}],""symbols"":[],""media"":[{""id"":494511934629093378,""id_str"":""494511934629093378"",""indices"":[83,105],""media_url"":""http://pbs.twimg.com/media/Btzb_AjCIAI0cLr.jpg"",""media_url_https"":""https://pbs.twimg.com/media/Btzb_AjCIAI0cLr.jpg"",""url"":""http://t.co/oC3DEK5cP6"",""display_url"":""pic.twitter.com/oC3DEK5cP6"",""expanded_url"":""http://twitter.com/xavijam/status/494511938563342336/photo/1"",""type"":""photo"",""sizes"":{""small"":{""w"":340,""h"":255,""resize"":""fit""},""medium"":{""w"":600,""h"":450,""resize"":""fit""},""large"":{""w"":1024,""h"":768,""resize"":""fit""},""thumb"":{""w"":150,""h"":150,""resize"":""crop""}}}]}","note","object:search.twitter.com,2005:494511938563342336",". @Javier is giving birth last @cartodb feature. All I can say is … GIF over maps! http://t.co/oC3DEK5cP6","2014-07-30T15:56:48.000Z","http://twitter.com/xavijam/statuses/494511938563342336","place","Trafalgar","https://api.twitter.com/1.1/geo/id/0144b1172069289c.json","{""type"":""Polygon"",""coordinates"":[[[-3.70582,40.4275708],[-3.70582,40.4386747],[-3.6956326,40.4386747],[-3.6956326,40.4275708]]]}",,"Trafalgar","{""urls"":[{""url"":""http://t.co/oC3DEK5cP6"",""expanded_url"":""http://twitter.com/xavijam/status/494511938563342336/photo/1"",""expanded_status"":200}],""klout_score"":39,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-3.70582,40.4275708],[-3.70582,40.4386747],[-3.6956326,40.4386747],[-3.6956326,40.4275708]]]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:494485754848894978","post","http://twitter.com/SonOfJorEl/statuses/494485754848894978","@cartoDB I’ve emailed support (which you claim has a 24hr turnaround) with minimal response. I am a paying customer ($149 mo). For how long?","activity","2014-07-30T14:12:45.000Z","0","medium","en","0","person","id:twitter.com:15628310","http://www.twitter.com/SonOfJorEl","Joey Schluchter","https://pbs.twimg.com/profile_images/466697750013562880/ngwKpO1j_normal.jpeg","Recovering golfer. Helping feed the world to software. Still love persimmon.","2008-07-28T05:34:03.000Z","[{""href"":""http://www.exzeo.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Clearwater, FL""}","-14400","SonOfJorEl","[""en""]","Eastern Time (US & Canada)","368","2053","5","2142","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[0,8]}],""symbols"":[]}","note","object:search.twitter.com,2005:494485754848894978","@cartoDB I’ve emailed support (which you claim has a 24hr turnaround) with minimal response. I am a paying customer ($149 mo). For how long?","2014-07-30T14:12:45.000Z","http://twitter.com/SonOfJorEl/statuses/494485754848894978","place","Tampa, FL","https://api.twitter.com/1.1/geo/id/dc62519fda13b4ec.json","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}",,"Tampa","{""klout_score"":42,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:494276921765543936","post","http://twitter.com/httsan/statuses/494276921765543936","""@gisuser: Online Mapping for Beginners – The Map Academy from CartoDB: TweetSomething innovative from… http://t.co/Zhh882TV2F""","activity","2014-07-30T00:22:56.000Z","0","medium","en","0","person","id:twitter.com:28471178","http://www.twitter.com/httsan","hartanto","https://pbs.twimg.com/profile_images/2217327231/IMG0033_normal.jpg","Lahir di Indonesia | Nguli di @NEOnetBPPT @BPPTeknologi | Ngobrol di @RSGISForum @PadangZaitun | Nyantri di @MichiganStateU","2009-04-03T01:32:35.000Z","[{""href"":""http://about.me/httsan"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""East Lansing""}","-14400","httsan","[""en""]","Eastern Time (US & Canada)","976","873","12","25725","false","Twitter for Android","http://twitter.com/download/android","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[42.7539688,-84.4221123]}","{""hashtags"":[],""symbols"":[],""urls"":[{""url"":""http://t.co/Zhh882TV2F"",""expanded_url"":""http://goo.gl/fb/xVB5rr"",""display_url"":""goo.gl/fb/xVB5rr"",""indices"":[104,126]}],""user_mentions"":[{""screen_name"":""gisuser"",""name"":""GISuser GeoTech News"",""id"":16958615,""id_str"":""16958615"",""indices"":[1,9]}]}","note","object:search.twitter.com,2005:494276921765543936","""@gisuser: Online Mapping for Beginners – The Map Academy from CartoDB: TweetSomething innovative from… http://t.co/Zhh882TV2F""","2014-07-30T00:22:56.000Z","http://twitter.com/httsan/statuses/494276921765543936","place","Haslett, MI","https://api.twitter.com/1.1/geo/id/4399b5004a3b4d9a.json","{""type"":""Polygon"",""coordinates"":[[[-84.447506,42.731229],[-84.447506,42.769688],[-84.363432,42.769688],[-84.363432,42.731229]]]}",,"Haslett","{""urls"":[{""url"":""http://t.co/Zhh882TV2F"",""expanded_url"":""http://blog.gisuser.com/2014/07/29/online-mapping-for-beginners-the-map-academy-from-cartodb/?utm_source=feedburner&utm_medium=twitter&utm_campaign=Feed:%20gisuser%20(GISUser.com%20-%20GIS,%20Mapping,%20Geospatial,%20and%20location%20technology%20news)"",""expanded_status"":200}],""klout_score"":51,""language"":{""value"":""en""}}","{""type"":""Point"",""coordinates"":[-84.4221123,42.7539688]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:494274906528288769","post","http://twitter.com/carygeo/statuses/494274906528288769","Attention mappers: iframe, #javascript, app development examples from the #CartoDB API/team! Well done @andrewxhill. http://t.co/cJM5fzgoyL","activity","2014-07-30T00:14:55.000Z","0","medium","en","0","person","id:twitter.com:2641688154","http://www.twitter.com/carygeo","Cary Greenwood","https://pbs.twimg.com/profile_images/488501325697544192/gFMxHuLV_normal.jpeg","geography, maps, data visualization, ui/ux design","2014-07-14T01:35:22.000Z","[{""href"":null,""rel"":""me""}]",,,"carygeo","[""en""]",,"109","46","0","47","false","Twitter for iPhone","http://twitter.com/download/iphone","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[34.13846767,-118.3620715]}","{""hashtags"":[{""text"":""javascript"",""indices"":[27,38]},{""text"":""CartoDB"",""indices"":[74,82]}],""symbols"":[],""urls"":[{""url"":""http://t.co/cJM5fzgoyL"",""expanded_url"":""http://vimeo.com/m/101936210"",""display_url"":""vimeo.com/m/101936210"",""indices"":[117,139]}],""user_mentions"":[{""screen_name"":""andrewxhill"",""name"":""Andrew W Hill"",""id"":19893224,""id_str"":""19893224"",""indices"":[103,115]}]}","note","object:search.twitter.com,2005:494274906528288769","Attention mappers: iframe, #javascript, app development examples from the #CartoDB API/team! Well done @andrewxhill. http://t.co/cJM5fzgoyL","2014-07-30T00:14:55.000Z","http://twitter.com/carygeo/statuses/494274906528288769","place","Los Angeles, CA","https://api.twitter.com/1.1/geo/id/3b77caf94bfc81fe.json","{""type"":""Polygon"",""coordinates"":[[[-118.668404,33.704538],[-118.668404,34.330724],[-118.155409,34.330724],[-118.155409,33.704538]]]}",,"Los Angeles","{""urls"":[{""url"":""http://t.co/cJM5fzgoyL"",""expanded_url"":""http://vimeo.com/101936210"",""expanded_status"":200}],""klout_score"":31,""language"":{""value"":""en""}}","{""type"":""Point"",""coordinates"":[-118.3620715,34.13846767]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:494194150133481472","post","http://twitter.com/juanjeojeda/statuses/494194150133481472","@dig_geo_com Your welcome :-) Keep in mind that some parts of @cartoDB still doesn't work fine. We're waiting for update from our provider.","activity","2014-07-29T18:54:01.000Z","0","medium","en","0","person","id:twitter.com:12679112","http://www.twitter.com/juanjeojeda","Juanje Ojeda ","https://pbs.twimg.com/profile_images/46025702/juanje_normal.jpg","Canarian guy who likes the open knowledge, the freedom and learn things every day. #OpenSource #DevOps #PaaS #Canarias","2008-01-25T11:49:15.000Z","[{""href"":""http://about.me/juanje"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Las Palmas de Gran Canaria""}","3600","juanjeojeda","[""es""]","London","416","1006","86","17700","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com","http://twitter.com/dig_geo_com/statuses/494191759078211584",,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""dig_geo_com"",""name"":""digital-geography"",""id"":231016902,""id_str"":""231016902"",""indices"":[0,12]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[62,70]}],""symbols"":[]}","note","object:search.twitter.com,2005:494194150133481472","@dig_geo_com Your welcome :-) Keep in mind that some parts of @cartoDB still doesn't work fine. We're waiting for update from our provider.","2014-07-29T18:54:01.000Z","http://twitter.com/juanjeojeda/statuses/494194150133481472","place","Las Palmas de Gran Canaria, Las Palmas","https://api.twitter.com/1.1/geo/id/00ab08eef1b62e92.json","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}",,"Las Palmas de Gran Canaria","{""klout_score"":56,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}","Category 2","aaa, bbb"
|
||||
"tag:search.twitter.com,2005:494191583240388608","post","http://twitter.com/juanjeojeda/statuses/494191583240388608","@dig_geo_com Hi, @cartoDB blog is already back. We had some issues with our hosting provider. Sorry for the inconveniences.","activity","2014-07-29T18:43:49.000Z","0","medium","en","0","person","id:twitter.com:12679112","http://www.twitter.com/juanjeojeda","Juanje Ojeda ","https://pbs.twimg.com/profile_images/46025702/juanje_normal.jpg","Canarian guy who likes the open knowledge, the freedom and learn things every day. #OpenSource #DevOps #PaaS #Canarias","2008-01-25T11:49:15.000Z","[{""href"":""http://about.me/juanje"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Las Palmas de Gran Canaria""}","3600","juanjeojeda","[""es""]","London","416","1006","86","17698","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com","http://twitter.com/dig_geo_com/statuses/494181592685117442",,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""dig_geo_com"",""name"":""digital-geography"",""id"":231016902,""id_str"":""231016902"",""indices"":[0,12]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[17,25]}],""symbols"":[]}","note","object:search.twitter.com,2005:494191583240388608","@dig_geo_com Hi, @cartoDB blog is already back. We had some issues with our hosting provider. Sorry for the inconveniences.","2014-07-29T18:43:49.000Z","http://twitter.com/juanjeojeda/statuses/494191583240388608","place","Las Palmas de Gran Canaria, Las Palmas","https://api.twitter.com/1.1/geo/id/00ab08eef1b62e92.json","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}
|
||||
|
Can't render this file because it contains an unexpected character in line 40 and column 1807.
|
@@ -0,0 +1,51 @@
|
||||
require_relative '../../lib/datasources'
|
||||
require_relative '../doubles/json_to_csv_converter'
|
||||
|
||||
include CartoDB::Datasources
|
||||
|
||||
describe CSVFileDumper do
|
||||
|
||||
before(:each) do
|
||||
end
|
||||
|
||||
describe '#dumping data' do
|
||||
it 'tests streaming dump' do
|
||||
output_stream_name ='/tmp/csv_file_dumper_test.csv'
|
||||
|
||||
File.unlink(output_stream_name) if File.exists?(output_stream_name)
|
||||
|
||||
converter_mock = CartoDB::TwitterSearch::Doubles::JSONToCSVConverter.new
|
||||
|
||||
dumper = CSVFileDumper.new(converter_mock, false)
|
||||
|
||||
dumper.buffer_size=2
|
||||
|
||||
input_filename_1 = 'stream_input_1'
|
||||
input_filename_2 = 'stream_input_2'
|
||||
input_data1 = File.read(File.join(File.dirname(__FILE__), "../fixtures/#{input_filename_1}"))
|
||||
input_data2 = File.read(File.join(File.dirname(__FILE__), "../fixtures/#{input_filename_2}"))
|
||||
|
||||
names_list = [ input_filename_1, input_filename_2 ]
|
||||
|
||||
stream = File.open(output_stream_name, 'wb')
|
||||
|
||||
dumper.begin_dump(input_filename_1)
|
||||
dumper.dump(input_filename_1, [input_data1])
|
||||
dumper.end_dump(input_filename_1)
|
||||
dumper.begin_dump(input_filename_2)
|
||||
dumper.dump(input_filename_2, [input_data2])
|
||||
dumper.end_dump(input_filename_2)
|
||||
|
||||
dumper.merge_dumps_into_stream(names_list, stream)
|
||||
|
||||
stream.close
|
||||
|
||||
data = File.read(output_stream_name)
|
||||
data.should eq "\n#{input_data1}\n#{input_data2}\n"
|
||||
File.unlink(output_stream_name)
|
||||
|
||||
end
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
@@ -0,0 +1,103 @@
|
||||
require_relative '../../lib/datasources'
|
||||
require_relative '../doubles/organization'
|
||||
require_relative '../doubles/user'
|
||||
require_relative '../doubles/search_tweet'
|
||||
require_relative '../doubles/data_import'
|
||||
require_relative '../../../../lib/cartodb/logger'
|
||||
|
||||
include CartoDB::Datasources
|
||||
|
||||
describe Search::Twitter do
|
||||
|
||||
def get_config
|
||||
{
|
||||
'auth_required' => false,
|
||||
'username' => '',
|
||||
'password' => '',
|
||||
'search_url' => 'http://fakeurl.carto',
|
||||
}
|
||||
end
|
||||
|
||||
before(:each) do
|
||||
Typhoeus::Expectation.clear
|
||||
end
|
||||
|
||||
describe '#search' do
|
||||
it 'tests basic full search flow with streaming' do
|
||||
user_quota = 100
|
||||
user_mock = CartoDB::Datasources::Doubles::User.new({twitter_datasource_quota: user_quota})
|
||||
data_import_mock = CartoDB::Datasources::Doubles::DataImport.new(id: '123456789', service_item_id: '987654321')
|
||||
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
|
||||
|
||||
input_terms = terms_fixture
|
||||
input_dates = dates_fixture
|
||||
|
||||
Typhoeus.stub(/fakeurl\.carto/) do |request|
|
||||
accept = (request.options[:headers]||{})['Accept'] || 'application/json'
|
||||
format = accept.split(',').first
|
||||
|
||||
request.base_url.should eq 'http://fakeurl.carto'
|
||||
request.options[:params].key?(:pusblisher).should eq false
|
||||
body = data_from_file('sample_tweets_v2.json')
|
||||
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => format },
|
||||
body: body
|
||||
)
|
||||
end
|
||||
|
||||
twitter_datasource.send :audit_entry, CartoDB::Datasources::Doubles::SearchTweet
|
||||
twitter_datasource.data_import_item = data_import_mock
|
||||
stream_location = '/tmp/sample_tweets_v2.csv'
|
||||
File.unlink(stream_location) if File.exists?(stream_location)
|
||||
stream = File.open(stream_location, 'wb')
|
||||
twitter_datasource.stream_resource(::JSON.dump(
|
||||
{
|
||||
categories: input_terms[:categories],
|
||||
dates: input_dates[:dates]
|
||||
}
|
||||
), stream)
|
||||
stream.close
|
||||
|
||||
stored_data = data_from_file(stream_location, true)
|
||||
stored_data.should eq data_from_file('sample_tweets_v2.csv')
|
||||
File.unlink(stream_location)
|
||||
end
|
||||
end
|
||||
|
||||
protected
|
||||
|
||||
def terms_fixture
|
||||
{
|
||||
categories: [
|
||||
{
|
||||
category: '1',
|
||||
terms: ['carto']
|
||||
}
|
||||
]
|
||||
}
|
||||
end
|
||||
|
||||
def dates_fixture
|
||||
{
|
||||
dates: {
|
||||
fromDate: '2017-02-21',
|
||||
fromHour: '11',
|
||||
fromMin: '45',
|
||||
toDate: '2017-02-21',
|
||||
toHour: '12',
|
||||
toMin: '00'
|
||||
}
|
||||
}
|
||||
end
|
||||
|
||||
def data_from_file(filename, fullpath=false)
|
||||
if fullpath
|
||||
File.read(filename)
|
||||
else
|
||||
File.read(File.join(File.dirname(__FILE__), "../fixtures/#{filename}"))
|
||||
end
|
||||
end
|
||||
|
||||
end
|
||||
@@ -0,0 +1,521 @@
|
||||
require 'active_support/core_ext'
|
||||
|
||||
require_relative '../../lib/datasources'
|
||||
require_relative '../doubles/user'
|
||||
|
||||
include CartoDB::Datasources
|
||||
|
||||
describe Url::ArcGIS do
|
||||
|
||||
before(:all) do
|
||||
@url = 'http://myserver/arcgis/rest/services/MyFakeService/featurename'
|
||||
@invalid_url = 'http://myserver/mysite/rest/myfakefolder/MyFakeService/featurename'
|
||||
@user = CartoDB::Datasources::Doubles::User.new
|
||||
end
|
||||
|
||||
before(:each) do
|
||||
Typhoeus::Expectation.clear
|
||||
end
|
||||
|
||||
describe '#set_data_from' do
|
||||
it 'tests preparing the correct url from the one given from the UI' do
|
||||
invalid_1 = 'http://myserver/services/MyFakeService/featurename/MapServer'
|
||||
invalid_2 = 'myserver/services/MyFakeService/featurename/MapServer'
|
||||
|
||||
test1 = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer'
|
||||
test2 = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer/'
|
||||
test3 = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer/0'
|
||||
test4 = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer/0?'
|
||||
test5 = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer/0?q=blablabla'
|
||||
test6 = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer?q=blablabla'
|
||||
|
||||
valid_map = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer'
|
||||
valid_map_trailing_slash = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer/'
|
||||
valid_layer = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer/0'
|
||||
|
||||
# Should be treated as ok
|
||||
test7 = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/2314/'
|
||||
valid_7 = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/2314/'
|
||||
|
||||
arcgis = Url::ArcGIS.get_new(@user)
|
||||
|
||||
expect {
|
||||
arcgis.send(:sanitize_id, invalid_1)
|
||||
}.to raise_error InvalidInputDataError
|
||||
expect {
|
||||
arcgis.send(:sanitize_id, invalid_2)
|
||||
}.to raise_error InvalidInputDataError
|
||||
|
||||
arcgis.send(:sanitize_id, test1).should eq valid_map
|
||||
arcgis.send(:sanitize_id, test2).should eq valid_map_trailing_slash
|
||||
arcgis.send(:sanitize_id, test3).should eq valid_layer
|
||||
arcgis.send(:sanitize_id, test4).should eq valid_layer
|
||||
arcgis.send(:sanitize_id, test5).should eq valid_layer
|
||||
arcgis.send(:sanitize_id, test6).should eq valid_map
|
||||
arcgis.send(:sanitize_id, test7).should eq valid_7
|
||||
end
|
||||
end
|
||||
|
||||
describe '#get_resource_metadata' do
|
||||
it 'tests error scenarios' do
|
||||
arcgis = Url::ArcGIS.get_new(@user)
|
||||
|
||||
sub_id = '0'
|
||||
|
||||
# 'general http error (non-200)'
|
||||
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/layers/) do
|
||||
Typhoeus::Response.new(
|
||||
code: 400,
|
||||
headers: { 'Content-Type' => 'application/json' },
|
||||
body: ''
|
||||
)
|
||||
end
|
||||
|
||||
expect {
|
||||
arcgis.get_resource_metadata(@url)
|
||||
}.to raise_error DataDownloadError
|
||||
|
||||
# Stub layers request (so now works)
|
||||
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/layers/) do |request|
|
||||
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_layers.json"))
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => 'application/json' },
|
||||
body: ::JSON.dump(::JSON.parse(body))
|
||||
)
|
||||
end
|
||||
|
||||
layers_data = arcgis.get_resource_metadata(@url)
|
||||
layers_data_expected = {
|
||||
id: @url,
|
||||
subresources: [{
|
||||
id: "#{@url}/#{sub_id}",
|
||||
title: 'first layer'
|
||||
}]
|
||||
}
|
||||
layers_data.should eq layers_data_expected
|
||||
|
||||
# 'fields' part
|
||||
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/0/) do |request|
|
||||
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_metadata_minimal.json"))
|
||||
|
||||
body = ::JSON.parse(body)
|
||||
body.delete('fields')
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => 'application/json' },
|
||||
body: ::JSON.dump(body)
|
||||
)
|
||||
end
|
||||
|
||||
expect {
|
||||
arcgis.send(:get_subresource_metadata, @url, sub_id)
|
||||
}.to raise_error ResponseError
|
||||
|
||||
# Another required field
|
||||
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/0/) do |request|
|
||||
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_metadata_minimal.json"))
|
||||
|
||||
body = ::JSON.parse(body)
|
||||
body.delete('name')
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => 'application/json' },
|
||||
body: ::JSON.dump(body)
|
||||
)
|
||||
end
|
||||
|
||||
expect {
|
||||
arcgis.send(:get_subresource_metadata, @url, sub_id)
|
||||
}.to raise_error ResponseError
|
||||
|
||||
# Invalid ArcGIS version
|
||||
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/0/) do |request|
|
||||
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_metadata_minimal.json"))
|
||||
|
||||
body = ::JSON.parse(body)
|
||||
body['currentVersion'] = 9.0
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => 'application/json' },
|
||||
body: ::JSON.dump(body)
|
||||
)
|
||||
end
|
||||
|
||||
expect {
|
||||
arcgis.send(:get_subresource_metadata, @url, sub_id)
|
||||
}.to raise_error InvalidServiceError
|
||||
|
||||
# Invalid ArcGIS URL
|
||||
expect {
|
||||
arcgis.send(:get_resource_metadata, @invalid_url)
|
||||
}.to raise_error InvalidInputDataError
|
||||
end
|
||||
|
||||
it 'tests metadata retrieval' do
|
||||
arcgis = Url::ArcGIS.get_new(@user)
|
||||
|
||||
# Stub layers request (so now works)
|
||||
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/layers/) do |request|
|
||||
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_layers.json"))
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => 'application/json' },
|
||||
body: ::JSON.dump(::JSON.parse(body))
|
||||
)
|
||||
end
|
||||
|
||||
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/0/) do |request|
|
||||
accept = (request.options[:headers]||{})['Accept'] || 'application/json'
|
||||
format = accept.split(',').first
|
||||
|
||||
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_metadata_minimal.json"))
|
||||
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => format },
|
||||
body: body
|
||||
)
|
||||
end
|
||||
|
||||
expected_metadata = {
|
||||
:arcgis_version=>10.22,
|
||||
:name=>"Test Feature",
|
||||
:description=>"Sample metadata payload",
|
||||
:type=>"Feature Layer",
|
||||
:geometry_type=>"esriGeometryPolygon",
|
||||
:copyright=>"CartoDB",
|
||||
:fields=>[
|
||||
{
|
||||
:name => "OBJECTID",
|
||||
:type => "esriFieldTypeOID"
|
||||
},
|
||||
{
|
||||
:name => "NAME",
|
||||
:type => "esriFieldTypeString"
|
||||
}
|
||||
],
|
||||
:max_records_per_query=>1000,
|
||||
:supported_formats=>["JSON", "AMF"],
|
||||
:advanced_queries_supported=>true
|
||||
}
|
||||
|
||||
expected_metadata_response = {
|
||||
id: @url + '/0',
|
||||
title: 'Test Feature',
|
||||
url: nil,
|
||||
service: Url::ArcGIS::DATASOURCE_NAME,
|
||||
checksum: nil,
|
||||
size: 0,
|
||||
filename: 'test_feature.json'
|
||||
}
|
||||
|
||||
# Multi-resource scenario already tested above
|
||||
response = arcgis.get_resource_metadata(@url + '/0')
|
||||
|
||||
response.nil?.should be false
|
||||
arcgis.metadata.should eq expected_metadata
|
||||
|
||||
response.should eq expected_metadata_response
|
||||
end
|
||||
end
|
||||
|
||||
describe '#get_resource' do
|
||||
it 'tests the get_ids_list() private method with error scenarios' do
|
||||
arcgis = Url::ArcGIS.get_new(@user)
|
||||
|
||||
id = arcgis.send(:sanitize_id, @url)
|
||||
|
||||
# 'general http error (non-200)'
|
||||
Typhoeus.stub(/\/arcgis\/rest\//) do
|
||||
Typhoeus::Response.new(
|
||||
code: 400,
|
||||
headers: { 'Content-Type' => 'application/json' },
|
||||
body: ''
|
||||
)
|
||||
end
|
||||
|
||||
expect {
|
||||
arcgis.send(:get_ids_list, id)
|
||||
}.to raise_error DataDownloadError
|
||||
|
||||
# 'objectIds' not present
|
||||
Typhoeus::Expectation.clear
|
||||
Typhoeus.stub(/\/arcgis\/rest\//) do
|
||||
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_ids_list.json"))
|
||||
|
||||
body = ::JSON.parse(body)
|
||||
body.delete('objectIds')
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => 'application/json' },
|
||||
body: ::JSON.dump(body)
|
||||
)
|
||||
end
|
||||
|
||||
expect {
|
||||
arcgis.send(:get_ids_list, id)
|
||||
}.to raise_error ResponseError
|
||||
|
||||
# 'objectIds' empty
|
||||
Typhoeus::Expectation.clear
|
||||
Typhoeus.stub(/\/arcgis\/rest\//) do
|
||||
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_ids_list.json"))
|
||||
|
||||
body = ::JSON.parse(body)
|
||||
body['objectIds'] = []
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => 'application/json' },
|
||||
body: ::JSON.dump(body)
|
||||
)
|
||||
end
|
||||
|
||||
expect {
|
||||
arcgis.send(:get_ids_list, id)
|
||||
}.to raise_error ResponseError
|
||||
|
||||
end
|
||||
|
||||
it 'tests the get_ids_list() private method' do
|
||||
arcgis = Url::ArcGIS.get_new(@user)
|
||||
|
||||
id = arcgis.send(:sanitize_id, @url)
|
||||
|
||||
Typhoeus.stub(/\/arcgis\/rest\//) do |request|
|
||||
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_ids_list.json"))
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => 'application/json' },
|
||||
body: body
|
||||
)
|
||||
end
|
||||
|
||||
expected_ids = [1,2,3,4,5,6,7,8,9,10]
|
||||
|
||||
respose_ids = arcgis.send(:get_ids_list, id)
|
||||
|
||||
respose_ids.nil?.should be false
|
||||
respose_ids.should eq expected_ids
|
||||
end
|
||||
|
||||
it 'tests the get_ids_list() private method on out-of-order ids' do
|
||||
arcgis = Url::ArcGIS.get_new(@user)
|
||||
|
||||
id = arcgis.send(:sanitize_id, @url)
|
||||
|
||||
Typhoeus.stub(/\/arcgis\/rest\//) do |request|
|
||||
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_unordered_ids_list.json"))
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => 'application/json' },
|
||||
body: body
|
||||
)
|
||||
end
|
||||
|
||||
expected_ids = [1,2,3,4,5,6,7,8,9,10]
|
||||
|
||||
respose_ids = arcgis.send(:get_ids_list, id)
|
||||
|
||||
respose_ids.nil?.should be false
|
||||
respose_ids.should eq expected_ids
|
||||
end
|
||||
|
||||
it 'tests the get_by_ids() private method with error scenarios' do
|
||||
arcgis = Url::ArcGIS.get_new(@user)
|
||||
|
||||
id = arcgis.send(:sanitize_id, @url)
|
||||
|
||||
# Empty ids
|
||||
expect {
|
||||
arcgis.send(:get_by_ids, id, [], [{ key: 'value' }])
|
||||
}.to raise_error InvalidInputDataError
|
||||
|
||||
# Empty fields
|
||||
expect {
|
||||
arcgis.send(:get_by_ids, id, [1], [])
|
||||
}.to raise_error InvalidInputDataError
|
||||
|
||||
|
||||
# 'general http error (non-200)'
|
||||
Typhoeus.stub(/\/arcgis\/rest\//) do
|
||||
Typhoeus::Response.new(
|
||||
code: 400,
|
||||
headers: { 'Content-Type' => 'application/json' },
|
||||
body: ''
|
||||
)
|
||||
end
|
||||
|
||||
expect {
|
||||
arcgis.send(:get_by_ids, id, [1], [{ key: 'value' }])
|
||||
}.to raise_error DataDownloadError
|
||||
end
|
||||
|
||||
it 'tests the get_by_ids() private method' do
|
||||
arcgis = Url::ArcGIS.get_new(@user)
|
||||
|
||||
id = arcgis.send(:sanitize_id, @url)
|
||||
|
||||
Typhoeus.stub(/\/arcgis\/rest\//) do
|
||||
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_data_01.json"))
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => 'application/json' },
|
||||
body: body
|
||||
)
|
||||
end
|
||||
|
||||
expected_response_data = {
|
||||
geometryType: "esriGeometryPolygon",
|
||||
spatialReference: {
|
||||
"wkid" => 4326,
|
||||
"latestWkid" => 4326
|
||||
},
|
||||
fields: [
|
||||
{
|
||||
"name" => "OBJECTID",
|
||||
"type" => "esriFieldTypeOID",
|
||||
"alias" => "OBJECTID"
|
||||
},
|
||||
{
|
||||
"name" => "WDPAID",
|
||||
"type" => "esriFieldTypeInteger",
|
||||
"alias" => "WDPAID"
|
||||
},
|
||||
{
|
||||
"name" => "NAME",
|
||||
"type" => "esriFieldTypeString",
|
||||
"alias" => "NAME",
|
||||
"length" => 254
|
||||
}
|
||||
],
|
||||
features: [
|
||||
{"attributes"=>{"OBJECTID"=>1, "NAME"=>"Name of object 1"}, "geometry"=>{"fake"=>"geom"}},
|
||||
{"attributes"=>{"OBJECTID"=>2, "NAME"=>"Name of object 2"}, "geometry"=>{"fake"=>"geom"}},
|
||||
{"attributes"=>{"OBJECTID"=>3, "NAME"=>"Name of object 3"}, "geometry"=>{"fake"=>"geom"}}
|
||||
]
|
||||
}
|
||||
|
||||
ids_to_retrieve = [1,2,3]
|
||||
# WDPAID also present, but left on purpose untouched
|
||||
fields_to_retrieve = [{
|
||||
name: 'OBJECTID',
|
||||
type: 'esriFieldTypeOID'
|
||||
},
|
||||
{
|
||||
name: 'NAME',
|
||||
type: 'esriFieldTypeString'
|
||||
}]
|
||||
|
||||
response_data = arcgis.send(:get_by_ids, id, ids_to_retrieve, fields_to_retrieve)
|
||||
|
||||
response_data.nil?.should eq false
|
||||
response_data[:features].length.should eq 3
|
||||
|
||||
response_data.should eq expected_response_data
|
||||
|
||||
end
|
||||
|
||||
it 'tests retrieval of data' do
|
||||
arcgis = Url::ArcGIS.get_new(@user)
|
||||
|
||||
feature_names = []
|
||||
|
||||
# Layers request
|
||||
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/layers/) do |request|
|
||||
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_layers.json"))
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => 'application/json' },
|
||||
body: ::JSON.dump(::JSON.parse(body))
|
||||
)
|
||||
end
|
||||
|
||||
# Metadata of a layer
|
||||
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/0\?f=json/) do
|
||||
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_metadata_minimal.json"))
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => 'application/json' },
|
||||
body: body
|
||||
)
|
||||
end
|
||||
|
||||
# IDs list of a layer
|
||||
Typhoeus.stub(/\/arcgis\/rest\/(.*)query\?where=/) do
|
||||
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_ids_list_01.json"))
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => 'application/json' },
|
||||
body: body
|
||||
)
|
||||
end
|
||||
|
||||
Typhoeus.stub(/\/arcgis\/rest\/(.*)query$/) do |response|
|
||||
if response.options[:body][:objectIds].to_i == 1
|
||||
# First item fetch of a layer
|
||||
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_data_01.json"))
|
||||
body = ::JSON.parse(body)
|
||||
|
||||
feature_names.push body['features'][0]['attributes']['NAME']
|
||||
body['features'] = [ body['features'][0] ]
|
||||
else
|
||||
# Remaining items fetch of a layer, will not use :objectIds
|
||||
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_data_01.json"))
|
||||
body = ::JSON.parse(body)
|
||||
|
||||
feature_names.push body['features'][1]['attributes']['NAME']
|
||||
feature_names.push body['features'][2]['attributes']['NAME']
|
||||
body['features'] = [ body['features'][1], body['features'][2] ]
|
||||
end
|
||||
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => 'application/json' },
|
||||
body: ::JSON.dump(body)
|
||||
)
|
||||
end
|
||||
|
||||
# 1) Retrieve lists of layers
|
||||
metadata = arcgis.get_resource_metadata(@url)
|
||||
|
||||
# 2) Retrieve metadata of a specific layer (doesn't adds much, but to replicate flow)
|
||||
item_metadata = arcgis.get_resource_metadata(metadata[:subresources].first[:id])
|
||||
|
||||
item_metadata.nil?.should eq false
|
||||
# No more checks as will fail later if missing something
|
||||
|
||||
spatial_ref_expectation = {"wkid"=>4326, "latestWkid"=>4326}
|
||||
|
||||
initial_stream_data = arcgis.initial_stream(item_metadata[:id])
|
||||
|
||||
initial_stream_data.nil?.should eq false
|
||||
initial_stream_data = ::JSON.parse(initial_stream_data)
|
||||
initial_stream_data['geometryType'].should eq 'esriGeometryPolygon'
|
||||
initial_stream_data['spatialReference'].should eq spatial_ref_expectation
|
||||
initial_stream_data['fields'].count.should eq 3
|
||||
initial_stream_data['fields'][0]['name'].should eq 'OBJECTID'
|
||||
initial_stream_data['fields'][1]['name'].should eq 'WDPAID'
|
||||
initial_stream_data['fields'][2]['name'].should eq 'NAME'
|
||||
initial_stream_data['features'].count.should eq 1
|
||||
initial_stream_data['features'][0]['attributes'].nil?.should eq false
|
||||
initial_stream_data['features'][0]['attributes']['NAME'].should eq feature_names[0]
|
||||
initial_stream_data['features'][0]['geometry'].nil?.should eq false
|
||||
|
||||
streamed_data = arcgis.stream_resource(item_metadata[:id])
|
||||
|
||||
streamed_data.nil?.should eq false
|
||||
streamed_data = ::JSON.parse(streamed_data)
|
||||
streamed_data['geometryType'].should eq 'esriGeometryPolygon'
|
||||
streamed_data['spatialReference'].should eq spatial_ref_expectation
|
||||
streamed_data['fields'].count.should eq 3
|
||||
streamed_data['fields'][0]['name'].should eq 'OBJECTID'
|
||||
streamed_data['fields'][1]['name'].should eq 'WDPAID'
|
||||
streamed_data['fields'][2]['name'].should eq 'NAME'
|
||||
streamed_data['features'].count.should eq 2
|
||||
streamed_data['features'][0]['attributes']['NAME'].should eq feature_names[1]
|
||||
streamed_data['features'][1]['attributes']['NAME'].should eq feature_names[2]
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,31 @@
|
||||
require_relative '../../lib/datasources'
|
||||
require_relative '../doubles/user'
|
||||
|
||||
include CartoDB::Datasources
|
||||
|
||||
describe Url::Box do
|
||||
def get_config
|
||||
{
|
||||
'box_host' => '',
|
||||
'application_name' => '',
|
||||
'client_id' => '',
|
||||
'client_secret' => '',
|
||||
'callback_url' => ''
|
||||
}
|
||||
end
|
||||
|
||||
describe '#filters' do
|
||||
it 'test that filter sets correctly' do
|
||||
user_mock = CartoDB::Datasources::Doubles::User.new
|
||||
|
||||
box_provider = Url::Box.get_new(get_config, user_mock)
|
||||
|
||||
box_provider.filter.should eq nil
|
||||
|
||||
# Filter to 'documents'
|
||||
formats = ['csv', 'xls']
|
||||
box_provider.filter = formats
|
||||
box_provider.filter.should eq formats
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,47 @@
|
||||
require_relative '../../lib/datasources'
|
||||
require_relative '../doubles/user'
|
||||
|
||||
include CartoDB::Datasources
|
||||
|
||||
describe Url::Dropbox do
|
||||
|
||||
def get_config
|
||||
{
|
||||
'app_key' => '',
|
||||
'app_secret' => '',
|
||||
'callback_url' => ''
|
||||
}
|
||||
end #get_config
|
||||
|
||||
describe '#filters' do
|
||||
it 'test that filter options work correctly' do
|
||||
user_mock = CartoDB::Datasources::Doubles::User.new
|
||||
|
||||
dropbox_provider = Url::Dropbox.get_new(get_config, user_mock)
|
||||
|
||||
# No filter = all formats allowed
|
||||
filter = []
|
||||
Url::Dropbox::FORMATS_TO_SEARCH_QUERIES.each do |id, search_queries|
|
||||
search_queries.each do |search_query|
|
||||
filter = filter.push(search_query)
|
||||
end
|
||||
end
|
||||
dropbox_provider.filter.should eq filter
|
||||
|
||||
# Filter to 'documents'
|
||||
filter = []
|
||||
format_ids = [ Url::Dropbox::FORMAT_CSV, Url::Dropbox::FORMAT_EXCEL ]
|
||||
Url::Dropbox::FORMATS_TO_SEARCH_QUERIES.each do |id, search_queries|
|
||||
if format_ids.include?(id)
|
||||
search_queries.each do |search_query|
|
||||
filter = filter.push(search_query)
|
||||
end
|
||||
end
|
||||
end
|
||||
dropbox_provider.filter = format_ids
|
||||
dropbox_provider.filter.should eq filter
|
||||
end
|
||||
end #run
|
||||
|
||||
end
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
require_relative '../../../../spec/rspec_configuration'
|
||||
require_relative '../../lib/datasources'
|
||||
require_relative '../doubles/user'
|
||||
|
||||
include CartoDB::Datasources
|
||||
|
||||
describe Url::GDrive do
|
||||
|
||||
def get_config
|
||||
{
|
||||
'application_name' => '',
|
||||
'client_id' => '',
|
||||
'client_secret' => '',
|
||||
'callback_url' => 'http://localhost/callback'
|
||||
}
|
||||
end #get_config
|
||||
|
||||
describe '#filters' do
|
||||
it 'test that filter options work correctly' do
|
||||
# Stubs Google Drive client for connectionless testing
|
||||
Google::Apis::DriveV2::DriveService.any_instance.stubs(:get_file)
|
||||
Google::Apis::DriveV2::DriveService.any_instance.stubs(:export_file)
|
||||
Google::Apis::DriveV2::DriveService.any_instance.stubs(:list_files)
|
||||
user_mock = CartoDB::Datasources::Doubles::User.new
|
||||
|
||||
gdrive_provider = Url::GDrive.get_new(get_config, user_mock)
|
||||
|
||||
# No filter = all formats allowed
|
||||
filter = []
|
||||
Url::GDrive::FORMATS_TO_MIME_TYPES.each do |id, mime_types|
|
||||
mime_types.each do |mime_type|
|
||||
filter = filter.push(mime_type)
|
||||
end
|
||||
end
|
||||
gdrive_provider.filter.should eq filter
|
||||
|
||||
# Filter to 'documents'
|
||||
filter = []
|
||||
format_ids = [ Url::GDrive::FORMAT_CSV, Url::GDrive::FORMAT_EXCEL ]
|
||||
Url::GDrive::FORMATS_TO_MIME_TYPES.each do |id, mime_types|
|
||||
if format_ids.include?(id)
|
||||
mime_types.each do |mime_type|
|
||||
filter = filter.push(mime_type)
|
||||
end
|
||||
end
|
||||
|
||||
end
|
||||
gdrive_provider.filter = format_ids
|
||||
gdrive_provider.filter.should eq filter
|
||||
end
|
||||
end #run
|
||||
|
||||
end
|
||||
@@ -0,0 +1,372 @@
|
||||
require_relative '../../lib/datasources'
|
||||
require_relative '../doubles/organization'
|
||||
require_relative '../doubles/user'
|
||||
require_relative '../doubles/search_tweet'
|
||||
require_relative '../doubles/data_import'
|
||||
require_relative '../../../../lib/cartodb/logger'
|
||||
|
||||
include CartoDB::Datasources
|
||||
|
||||
describe Search::Twitter do
|
||||
|
||||
def get_config
|
||||
{
|
||||
'auth_required' => false,
|
||||
'username' => '',
|
||||
'password' => '',
|
||||
'search_url' => 'http://fakeurl.carto',
|
||||
}
|
||||
end #get_config
|
||||
|
||||
before(:each) do
|
||||
Typhoeus::Expectation.clear
|
||||
end
|
||||
|
||||
describe '#filters' do
|
||||
it 'tests max and total results filters' do
|
||||
big_quota = 123456
|
||||
user = CartoDB::Datasources::Doubles::User.new({
|
||||
twitter_datasource_quota: big_quota
|
||||
})
|
||||
twitter_datasource = Search::Twitter.get_new(get_config, user)
|
||||
|
||||
maxresults_filter = twitter_datasource.send :build_maxresults_field, user
|
||||
maxresults_filter.should eq CartoDB::TwitterSearch::SearchAPI::MAX_PAGE_RESULTS
|
||||
totalresults_filter = twitter_datasource.send :build_total_results_field, user
|
||||
totalresults_filter.should eq big_quota
|
||||
|
||||
small_quota = 13
|
||||
user = CartoDB::Datasources::Doubles::User.new({
|
||||
twitter_datasource_quota: small_quota,
|
||||
soft_twitter_datasource_limit: false
|
||||
})
|
||||
twitter_datasource = Search::Twitter.get_new(get_config, user)
|
||||
maxresults_filter = twitter_datasource.send :build_maxresults_field, user
|
||||
maxresults_filter.should eq small_quota
|
||||
totalresults_filter = twitter_datasource.send :build_total_results_field, user
|
||||
totalresults_filter.should eq small_quota
|
||||
|
||||
user = CartoDB::Datasources::Doubles::User.new({
|
||||
twitter_datasource_quota: small_quota,
|
||||
soft_twitter_datasource_limit: true
|
||||
})
|
||||
maxresults_filter = twitter_datasource.send :build_maxresults_field, user
|
||||
maxresults_filter.should eq CartoDB::TwitterSearch::SearchAPI::MAX_PAGE_RESULTS
|
||||
totalresults_filter = twitter_datasource.send :build_total_results_field, user
|
||||
totalresults_filter.should eq Search::Twitter::NO_TOTAL_RESULTS
|
||||
end
|
||||
|
||||
it 'tests category filters' do
|
||||
user_mock = CartoDB::Datasources::Doubles::User.new
|
||||
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
|
||||
|
||||
input_terms = terms_fixture
|
||||
|
||||
expected_output_terms = [
|
||||
{
|
||||
Search::Twitter::CATEGORY_NAME_KEY => 'Category 1',
|
||||
Search::Twitter::CATEGORY_TERMS_KEY => '(uno OR @dos OR #tres) (has:geo OR has:profile_geo)'
|
||||
},
|
||||
{
|
||||
Search::Twitter::CATEGORY_NAME_KEY => 'Category 2',
|
||||
Search::Twitter::CATEGORY_TERMS_KEY => '(aaa OR bbb) (has:geo OR has:profile_geo)'
|
||||
}
|
||||
]
|
||||
|
||||
output = twitter_datasource.send :build_queries_from_fields, input_terms
|
||||
|
||||
output.should eq expected_output_terms
|
||||
end
|
||||
|
||||
it 'tests search term cut if too many' do
|
||||
user_mock = CartoDB::Datasources::Doubles::User.new
|
||||
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
|
||||
|
||||
input_terms = {
|
||||
categories: [
|
||||
{
|
||||
category: 'Category 1',
|
||||
terms: Array(1..35)
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
expected_output_terms = [
|
||||
{
|
||||
Search::Twitter::CATEGORY_NAME_KEY => 'Category 1',
|
||||
Search::Twitter::CATEGORY_TERMS_KEY => '(1 OR 2 OR 3 OR 4 OR 5 OR 6 OR 7 OR 8 OR 9 OR 10 OR 11 OR 12 OR 13 OR 14 OR 15 OR 16 OR 17 OR 18 OR 19 OR 20 OR 21 OR 22 OR 23 OR 24 OR 25 OR 26 OR 27 OR 28 OR 29) (has:geo OR has:profile_geo)'
|
||||
},
|
||||
]
|
||||
|
||||
output = twitter_datasource.send :build_queries_from_fields, input_terms
|
||||
output.should eq expected_output_terms
|
||||
end
|
||||
|
||||
it 'tests search term cut if too big (even if amount is ok)' do
|
||||
user_mock = CartoDB::Datasources::Doubles::User.new
|
||||
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
|
||||
|
||||
input_terms = {
|
||||
categories: [
|
||||
{
|
||||
category: 'Category 1',
|
||||
terms: ['wadus1', 'wadus2', 'wadus3' * 500]
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
expect {
|
||||
output = twitter_datasource.send :build_queries_from_fields, input_terms
|
||||
|
||||
}.to raise_error ParameterError
|
||||
end
|
||||
|
||||
|
||||
it 'tests date filters' do
|
||||
user_mock = CartoDB::Datasources::Doubles::User.new
|
||||
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
|
||||
|
||||
input_dates = dates_fixture
|
||||
|
||||
output = twitter_datasource.send :build_date_from_fields, input_dates, 'from'
|
||||
output.should eq '201403031349'
|
||||
|
||||
output = twitter_datasource.send :build_date_from_fields, input_dates, 'to'
|
||||
output.should eq '201403041159'
|
||||
|
||||
expect {
|
||||
twitter_datasource.send :build_date_from_fields, input_dates, 'wadus'
|
||||
}.to raise_error ParameterError
|
||||
|
||||
|
||||
current_time = Time.now.utc
|
||||
output = twitter_datasource.send :build_date_from_fields, {
|
||||
dates: {
|
||||
toDate: current_time.strftime("%Y-%m-%d"),
|
||||
toHour: current_time.hour + 1, # Set into the future
|
||||
toMin: current_time.min
|
||||
}
|
||||
}, 'to'
|
||||
output.should eq nil
|
||||
|
||||
end
|
||||
|
||||
it 'tests twitter search integration (without conversion to CSV)' do
|
||||
# This test bridges lots of internal calls to simulate only up until twitter search call and results
|
||||
user_mock = CartoDB::Datasources::Doubles::User.new
|
||||
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
|
||||
|
||||
input_terms = terms_fixture
|
||||
input_dates = dates_fixture
|
||||
|
||||
Typhoeus.stub(/fakeurl\.carto/) do |request|
|
||||
accept = (request.options[:headers]||{})['Accept'] || 'application/json'
|
||||
format = accept.split(',').first
|
||||
|
||||
if request.options[:params][:next].nil?
|
||||
body = data_from_file('sample_tweets.json')
|
||||
else
|
||||
body = data_from_file('sample_tweets_2.json')
|
||||
end
|
||||
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => format },
|
||||
body: body
|
||||
)
|
||||
end
|
||||
|
||||
twitter_api_config = twitter_datasource.send :search_api_config
|
||||
twitter_api = CartoDB::TwitterSearch::SearchAPI.new(twitter_api_config)
|
||||
|
||||
fields = {
|
||||
categories: input_terms[:categories],
|
||||
dates: input_dates[:dates]
|
||||
}
|
||||
filters = {
|
||||
Search::Twitter::FILTER_CATEGORIES => (twitter_datasource.send :build_queries_from_fields, fields),
|
||||
Search::Twitter::FILTER_FROMDATE => (twitter_datasource.send :build_date_from_fields, fields, 'from'),
|
||||
Search::Twitter::FILTER_TODATE => (twitter_datasource.send :build_date_from_fields, fields, 'to'),
|
||||
Search::Twitter::FILTER_MAXRESULTS => 500,
|
||||
Search::Twitter::FILTER_TOTAL_RESULTS => Search::Twitter::NO_TOTAL_RESULTS
|
||||
}
|
||||
|
||||
category = {
|
||||
name: input_terms[:categories].first[:category],
|
||||
terms: input_terms[:categories].first[:terms],
|
||||
}
|
||||
csv_dumper = twitter_datasource.send :csv_dumper
|
||||
csv_dumper.begin_dump(input_terms[:categories][0][:category])
|
||||
csv_dumper.begin_dump(input_terms[:categories][1][:category])
|
||||
csv_dumper.additional_fields = { category[:name] => category }
|
||||
|
||||
output = twitter_datasource.send :search_by_category, twitter_api, filters, category
|
||||
|
||||
# 2 pages of 10 results per category search
|
||||
output.should eq 20
|
||||
|
||||
csv_dumper.send :destroy_files
|
||||
end
|
||||
|
||||
it 'tests stopping search if runs out of quota' do
|
||||
# Should equal to sample_tweets_3.json number of results, and always >= 10 (because is Gnip's minimum)
|
||||
remaining_tweets_quota = 11
|
||||
|
||||
user_mock = CartoDB::Datasources::Doubles::User.new(
|
||||
twitter_datasource_quota: remaining_tweets_quota
|
||||
)
|
||||
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
|
||||
|
||||
input_terms = terms_fixture
|
||||
input_dates = dates_fixture
|
||||
|
||||
Typhoeus.stub(/fakeurl\.carto/) do |request|
|
||||
accept = (request.options[:headers]||{})['Accept'] || 'application/json'
|
||||
format = accept.split(',').first
|
||||
|
||||
if request.options[:params][:next].nil?
|
||||
# This dataset has 11 items and a "next"
|
||||
body = data_from_file('sample_tweets_3.json')
|
||||
else
|
||||
body = data_from_file('sample_tweets_2.json')
|
||||
end
|
||||
|
||||
Typhoeus::Response.new(
|
||||
code: 200,
|
||||
headers: { 'Content-Type' => format },
|
||||
body: body
|
||||
)
|
||||
end
|
||||
|
||||
twitter_api_config = twitter_datasource.send :search_api_config
|
||||
twitter_api = CartoDB::TwitterSearch::SearchAPI.new(twitter_api_config)
|
||||
|
||||
fields = {
|
||||
categories: input_terms[:categories],
|
||||
dates: input_dates[:dates]
|
||||
}
|
||||
filters = {
|
||||
Search::Twitter::FILTER_CATEGORIES => (twitter_datasource.send :build_queries_from_fields, fields),
|
||||
Search::Twitter::FILTER_FROMDATE => (twitter_datasource.send :build_date_from_fields, fields, 'from'),
|
||||
Search::Twitter::FILTER_TODATE => (twitter_datasource.send :build_date_from_fields, fields, 'to'),
|
||||
Search::Twitter::FILTER_MAXRESULTS => 500,
|
||||
Search::Twitter::FILTER_TOTAL_RESULTS => Search::Twitter::NO_TOTAL_RESULTS
|
||||
}
|
||||
|
||||
category = {
|
||||
name: input_terms[:categories].first[:category],
|
||||
terms: input_terms[:categories].first[:terms],
|
||||
}
|
||||
csv_dumper = twitter_datasource.send :csv_dumper
|
||||
csv_dumper.begin_dump(input_terms[:categories][0][:category])
|
||||
csv_dumper.begin_dump(input_terms[:categories][1][:category])
|
||||
csv_dumper.additional_fields = { category[:name] => category }
|
||||
|
||||
output = twitter_datasource.send :search_by_category, twitter_api, filters, category
|
||||
|
||||
output.should eq remaining_tweets_quota
|
||||
|
||||
csv_dumper.send :destroy_files
|
||||
end
|
||||
|
||||
it 'tests user limits on datasource usage' do
|
||||
user_mock = CartoDB::Datasources::Doubles::User.new
|
||||
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
|
||||
|
||||
# Service enabled tests
|
||||
result = twitter_datasource.send :is_service_enabled?, CartoDB::Datasources::Doubles::User.new({
|
||||
has_org: true,
|
||||
twitter_datasource_enabled: true,
|
||||
org_twitter_datasource_enabled: true
|
||||
})
|
||||
result.should eq true
|
||||
|
||||
result = twitter_datasource.send :is_service_enabled?, CartoDB::Datasources::Doubles::User.new({
|
||||
has_org: true,
|
||||
twitter_datasource_enabled: false,
|
||||
org_twitter_datasource_enabled: true
|
||||
})
|
||||
result.should eq false
|
||||
|
||||
result = twitter_datasource.send :is_service_enabled?, CartoDB::Datasources::Doubles::User.new({
|
||||
twitter_datasource_enabled: true,
|
||||
})
|
||||
result.should eq true
|
||||
|
||||
result = twitter_datasource.send :is_service_enabled?, CartoDB::Datasources::Doubles::User.new({
|
||||
twitter_datasource_enabled: false,
|
||||
})
|
||||
result.should eq false
|
||||
|
||||
# Quota & soft limit tests
|
||||
result = twitter_datasource.send :has_enough_quota?, CartoDB::Datasources::Doubles::User.new({
|
||||
soft_twitter_datasource_limit: false,
|
||||
twitter_datasource_quota: 10,
|
||||
})
|
||||
result.should eq true
|
||||
|
||||
result = twitter_datasource.send :has_enough_quota?, CartoDB::Datasources::Doubles::User.new({
|
||||
soft_twitter_datasource_limit: true,
|
||||
twitter_datasource_quota: 0,
|
||||
})
|
||||
result.should eq true
|
||||
|
||||
result = twitter_datasource.send :has_enough_quota?, CartoDB::Datasources::Doubles::User.new({
|
||||
soft_twitter_datasource_limit: false,
|
||||
twitter_datasource_quota: 0,
|
||||
})
|
||||
result.should eq false
|
||||
end
|
||||
|
||||
it 'checks terms sanitize method' do
|
||||
user_mock = CartoDB::Datasources::Doubles::User.new
|
||||
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
|
||||
|
||||
terms = [ 'a', ' b', 'c ', ' d ', ' e f', 'g h ', ' i j ', ' 1 2 3 4 ', ' ' ]
|
||||
terms_expected = [ 'a', 'b', 'c', 'd', '"e f"', '"g h"', '"i j"', '"1 2 3 4"' ]
|
||||
|
||||
result = twitter_datasource.send :sanitize_terms, terms
|
||||
result.should eq terms_expected
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
protected
|
||||
|
||||
def terms_fixture
|
||||
{
|
||||
categories: [
|
||||
{
|
||||
category: 'Category 1',
|
||||
terms: ['uno', '@dos', '#tres']
|
||||
},
|
||||
{
|
||||
category: 'Category 2',
|
||||
terms: ['aaa', 'bbb']
|
||||
}
|
||||
]
|
||||
}
|
||||
end
|
||||
|
||||
def dates_fixture
|
||||
{
|
||||
dates: {
|
||||
fromDate: '2014-03-03',
|
||||
fromHour: '13',
|
||||
fromMin: '49',
|
||||
toDate: '2014-03-04',
|
||||
toHour: '11',
|
||||
toMin: '59'
|
||||
}
|
||||
}
|
||||
end
|
||||
|
||||
def data_from_file(filename, fullpath=false)
|
||||
if fullpath
|
||||
File.read(filename)
|
||||
else
|
||||
File.read(File.join(File.dirname(__FILE__), "../fixtures/#{filename}"))
|
||||
end
|
||||
end
|
||||
|
||||
end
|
||||
@@ -0,0 +1,16 @@
|
||||
require 'singleton'
|
||||
|
||||
module CartoDB
|
||||
# A facility that abstracts clients from config and also allow for easy injection
|
||||
class GeocoderConfig
|
||||
include Singleton
|
||||
|
||||
def set(config = {})
|
||||
@config = config
|
||||
end
|
||||
|
||||
def get()
|
||||
@config ||= ::Cartodb.config[:geocoder]
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,320 @@
|
||||
require 'open3'
|
||||
require 'nokogiri'
|
||||
require 'csv'
|
||||
require 'active_support/core_ext/numeric'
|
||||
require_relative '../../../lib/carto/http/client'
|
||||
require_relative 'hires_geocoder_interface'
|
||||
require_relative 'geocoder_config'
|
||||
|
||||
module CartoDB
|
||||
class HiresBatchGeocoder < HiresGeocoderInterface
|
||||
|
||||
DEFAULT_TIMEOUT = 5.hours
|
||||
POLLING_SLEEP_TIME = 5.seconds
|
||||
LOGGING_TIME = 5.minutes
|
||||
DOWNLOAD_RETRIES = 5
|
||||
DOWLOAD_RETRY_SLEEP = 5.seconds
|
||||
|
||||
# Generous timeouts, overriden for big files upload/download
|
||||
HTTP_CONNECTION_TIMEOUT = 60
|
||||
HTTP_REQUEST_TIMEOUT = 600
|
||||
|
||||
# Options for the csv upload endpoint of the Batch Geocoder API
|
||||
UPLOAD_OPTIONS = {
|
||||
action: 'run',
|
||||
indelim: ',',
|
||||
outdelim: ',',
|
||||
header: false,
|
||||
outputCombined: false,
|
||||
outcols: "displayLatitude,displayLongitude"
|
||||
}
|
||||
|
||||
# INFO: the request_id is the most important thing to care for batch requests
|
||||
# INFO: it is called remote_id in upper layers
|
||||
attr_reader :base_url, :request_id, :app_id, :token, :mailto,
|
||||
:status, :processed_rows, :processed_rows, :successful_processed_rows, :failed_processed_rows,
|
||||
:empty_processed_rows, :total_rows, :dir, :input_file
|
||||
|
||||
class ServiceDisabled < StandardError; end
|
||||
|
||||
|
||||
def initialize(input_csv_file, working_dir, log, geocoding_model)
|
||||
@input_file = input_csv_file
|
||||
@dir = working_dir
|
||||
@log = log
|
||||
@geocoding_model = geocoding_model
|
||||
@base_url = config.fetch('base_url')
|
||||
@app_id = config.fetch('app_id')
|
||||
@token = config.fetch('token')
|
||||
@mailto = config.fetch('mailto')
|
||||
@used_batch_request = true
|
||||
begin
|
||||
@batch_api_disabled = config['batch_api_disabled'] == true
|
||||
rescue
|
||||
@batch_api_disabled = false
|
||||
end
|
||||
end
|
||||
|
||||
def run
|
||||
init_rows_count
|
||||
@log.append_and_store "Started batched Here geocoding job"
|
||||
@started_at = Time.now
|
||||
change_status('running')
|
||||
upload
|
||||
|
||||
# INFO: this loop polls for the state of the table_geocoder batch process
|
||||
update_status
|
||||
until ['completed', 'cancelled'].include? @geocoding_model.state do
|
||||
if timeout?
|
||||
begin
|
||||
change_status('timeout')
|
||||
cancel
|
||||
ensure
|
||||
@log.append_and_store "Proceding to cancel job due timeout"
|
||||
end
|
||||
end
|
||||
|
||||
break if ['failed', 'timeout'].include? @geocoding_model.state
|
||||
|
||||
sleep polling_sleep_time
|
||||
# We don't want to change the status if the job has been cancelled by the user
|
||||
update_status
|
||||
update_log_stats
|
||||
end
|
||||
update_status
|
||||
update_log_stats
|
||||
change_status('completed')
|
||||
@log.append_and_store "Geocoding Hires job has finished"
|
||||
ensure
|
||||
# Processed data at the end of the job
|
||||
update_status
|
||||
update_log_stats(false)
|
||||
end
|
||||
|
||||
def upload
|
||||
assert_batch_api_enabled
|
||||
@used_batch_request = true
|
||||
response = http_client.post(
|
||||
api_url(UPLOAD_OPTIONS),
|
||||
body: File.open(input_file, "r").read,
|
||||
headers: { "Content-Type" => "text/plain" },
|
||||
timeout: 5.hours # more than generous timeout for big file upload
|
||||
)
|
||||
handle_api_error(response)
|
||||
@request_id = extract_response_field(response.body, '//Response/MetaInfo/RequestId')
|
||||
# TODO: this is a critical error, deal with it appropriately
|
||||
raise 'Could not get the request ID' unless @request_id
|
||||
# Update geocodings model with needed data
|
||||
@geocoding_model.remote_id = @request_id
|
||||
@geocoding_model.batched = true
|
||||
@geocoding_model.save
|
||||
@log.append_and_store "Job sent to HERE, job id: #{@request_id}"
|
||||
|
||||
@request_id
|
||||
end
|
||||
|
||||
def used_batch_request?
|
||||
@used_batch_request
|
||||
end
|
||||
|
||||
def cancel
|
||||
if @geocoding_model.remote_id.nil?
|
||||
@log.append_and_store "Can't cancel a HERE geocoder job without the request id"
|
||||
else
|
||||
@log.append_and_store "Trying to cancel a batch job sent to HERE"
|
||||
assert_batch_api_enabled
|
||||
response = http_client.put(api_url(action: 'cancel'),
|
||||
connecttimeout: HTTP_CONNECTION_TIMEOUT,
|
||||
timeout: HTTP_REQUEST_TIMEOUT)
|
||||
if is_cancellable?(response)
|
||||
@log.append_and_store "Job was already cancelled"
|
||||
else
|
||||
handle_api_error(response)
|
||||
update_stats(response)
|
||||
@log.append_and_store "Job sent to HERE has been cancelled"
|
||||
end
|
||||
change_status('cancelled')
|
||||
end
|
||||
end
|
||||
|
||||
def update_status
|
||||
assert_batch_api_enabled
|
||||
response = http_client.get(api_url(action: 'status'),
|
||||
connecttimeout: HTTP_CONNECTION_TIMEOUT,
|
||||
timeout: HTTP_REQUEST_TIMEOUT)
|
||||
handle_api_error(response)
|
||||
update_stats(response)
|
||||
end
|
||||
|
||||
def assert_batch_api_enabled
|
||||
raise ServiceDisabled if @batch_api_disabled
|
||||
end
|
||||
|
||||
def result
|
||||
return @result unless @result.nil?
|
||||
|
||||
raise 'No request_id provided' unless @geocoding_model.remote_id
|
||||
results_filename = File.join(dir, "#{@geocoding_model.remote_id}.zip")
|
||||
download_url = api_url({}, 'result')
|
||||
download_status_code = nil
|
||||
retries = 0
|
||||
while true
|
||||
if(!download_status_code.nil? && download_status_code == 200)
|
||||
break
|
||||
elsif !download_status_code.nil? && download_status_code == 404
|
||||
# 404 means that the results file is not ready yet
|
||||
sleep DOWLOAD_RETRY_SLEEP
|
||||
retries += 1
|
||||
elsif retries >= DOWNLOAD_RETRIES
|
||||
raise 'Download request failed: Too many retries, should be a problem with HERE servers'
|
||||
elsif !download_status_code.nil? && download_status_code > 200 && download_status_code != 404
|
||||
raise "Download request failed: Http status code #{download_status_code}"
|
||||
end
|
||||
download_status_code = execute_results_request(download_url, results_filename)
|
||||
end
|
||||
@result = results_filename
|
||||
end
|
||||
|
||||
|
||||
private
|
||||
|
||||
def execute_results_request(download_url, results_filename)
|
||||
download_status_code = nil
|
||||
# generous timeout for download of results
|
||||
request = http_client.request(download_url,
|
||||
method: :get,
|
||||
timeout: 5.hours)
|
||||
|
||||
File.open(results_filename, 'wb') do |download_file|
|
||||
request.on_headers do |response|
|
||||
download_status_code = response.response_code
|
||||
end
|
||||
|
||||
request.on_body do |chunk|
|
||||
if download_status_code == 200
|
||||
download_file.write(chunk)
|
||||
end
|
||||
end
|
||||
|
||||
request.on_complete do |response|
|
||||
download_status_code = response.response_code
|
||||
end
|
||||
|
||||
request.run
|
||||
end
|
||||
|
||||
return download_status_code
|
||||
end
|
||||
|
||||
def config
|
||||
GeocoderConfig.instance.get
|
||||
end
|
||||
|
||||
def http_client
|
||||
@http_client ||= Carto::Http::Client.get('hires_batch_geocoder',
|
||||
log_requests: true)
|
||||
end
|
||||
|
||||
def api_url(arguments, extra_components = nil)
|
||||
arguments.merge!(app_id: app_id, token: token, mailto: mailto)
|
||||
components = [base_url]
|
||||
# We use the persisted remote_id because we don't have request_id
|
||||
# in the cancel case due is an instance variable
|
||||
components << @geocoding_model.remote_id unless @geocoding_model.remote_id.nil?
|
||||
components << extra_components unless extra_components.nil?
|
||||
components << '?' + URI.encode_www_form(arguments)
|
||||
components.join('/')
|
||||
end
|
||||
|
||||
def extract_response_field(response, query)
|
||||
Nokogiri::XML(response).xpath("#{query}").first.content
|
||||
rescue NoMethodError => e
|
||||
CartoDB.notify_exception(e)
|
||||
nil
|
||||
end
|
||||
|
||||
def extract_numeric_response_field(response, query)
|
||||
value = extract_response_field(response, query)
|
||||
return nil if value.blank?
|
||||
Integer(value)
|
||||
rescue ArgumentError => e
|
||||
CartoDB.notify_error("Batch geocoder value error", error: e.message, value: value)
|
||||
nil
|
||||
end
|
||||
|
||||
def handle_api_error(response)
|
||||
if response.success? == false
|
||||
message = extract_response_field(response.body, '//Details')
|
||||
@failed_processed_rows = number_of_input_file_rows if not input_file.nil?
|
||||
change_status('failed')
|
||||
raise "Geocoding API communication failure: #{message}"
|
||||
end
|
||||
end
|
||||
|
||||
def default_timeout
|
||||
DEFAULT_TIMEOUT
|
||||
end
|
||||
|
||||
def polling_sleep_time
|
||||
POLLING_SLEEP_TIME
|
||||
end
|
||||
|
||||
def number_of_input_file_rows
|
||||
stdout, _status = Open3.capture2('wc', '-l', input_file)
|
||||
stdout.to_i
|
||||
end
|
||||
|
||||
def update_stats(response)
|
||||
@status = extract_response_field(response.body, '//Response/Status')
|
||||
change_status(@status)
|
||||
@processed_rows = extract_numeric_response_field(response.body, '//Response/ProcessedCount')
|
||||
@successful_processed_rows = extract_numeric_response_field(response.body, '//Response/SuccessCount')
|
||||
# addresses that could not be matched
|
||||
@empty_processed_rows = extract_numeric_response_field(response.body, '//Response/ErrorCount')
|
||||
# invalid input that could not be processed
|
||||
@failed_processed_rows = extract_numeric_response_field(response.body, '//Response/InvalidCount')
|
||||
@total_rows = extract_numeric_response_field(response.body, '//Response/TotalCount')
|
||||
end
|
||||
|
||||
def init_rows_count
|
||||
@processed_rows = 0
|
||||
@successful_processed_rows = 0
|
||||
@empty_processed_rows = 0
|
||||
@failed_processed_rows = 0
|
||||
@total_rows = 0
|
||||
end
|
||||
|
||||
def update_log_stats(spaced_by_time=true)
|
||||
@last_logging_time ||= Time.now
|
||||
# We don't want to log every few seconds because this kind
|
||||
# of jobs could last for hours
|
||||
if (not spaced_by_time) || (Time.now - @last_logging_time) > LOGGING_TIME
|
||||
@log.append_and_store "Geocoding job status update. "\
|
||||
"Status: #{@geocoding_model.state} --- Processed rows: #{@processed_rows} "\
|
||||
"--- Success: #{@successful_processed_rows} --- Empty: #{@empty_processed_rows} "\
|
||||
"--- Failed: #{@failed_processed_rows}"
|
||||
@last_logging_time = Time.now
|
||||
end
|
||||
end
|
||||
|
||||
def timeout?
|
||||
(Time.now - @started_at) > default_timeout
|
||||
end
|
||||
|
||||
def change_status(status)
|
||||
@status = status
|
||||
# The cancelled status should prevail to abort the job
|
||||
@geocoding_model.refresh
|
||||
if status != @geocoding_model.state && (not (@geocoding_model.cancelled? || @geocoding_model.timeout?))
|
||||
@geocoding_model.state = status
|
||||
@geocoding_model.save
|
||||
end
|
||||
end
|
||||
|
||||
def is_cancellable?(response)
|
||||
message = extract_response_field(response.body, '//Details')
|
||||
response.response_code == 400 && message =~ /CANNOT CANCEL THE COMPLETED, DELETED, FAILED OR ALREADY CANCELLED JOB/
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,169 @@
|
||||
require 'csv'
|
||||
require 'json'
|
||||
require 'open3'
|
||||
require_relative '../../../lib/carto/http/client'
|
||||
require_relative 'hires_geocoder_interface'
|
||||
require_relative 'geocoder_config'
|
||||
|
||||
module CartoDB
|
||||
class HiresGeocoder < HiresGeocoderInterface
|
||||
|
||||
# Generous timeouts for this
|
||||
HTTP_CONNECTION_TIMEOUT = 60
|
||||
HTTP_REQUEST_TIMEOUT = 600
|
||||
|
||||
# Default options for the regular HERE Geocoding API
|
||||
# Refer to developer.here.com for further reading
|
||||
GEOCODER_OPTIONS = {
|
||||
gen: 4, # enables or disables backward incompatible behavior in the API
|
||||
jsonattributes: 1, # lowercase the first character of each JSON response attribute name
|
||||
language: 'en-US', # preferred language of address elements in the result
|
||||
maxresults: 1
|
||||
}
|
||||
|
||||
attr_reader :app_id, :token, :mailto,
|
||||
:status, :processed_rows, :total_rows, :successful_processed_rows, :failed_processed_rows,
|
||||
:empty_processed_rows, :dir, :non_batch_base_url
|
||||
|
||||
attr_accessor :input_file
|
||||
|
||||
def initialize(input_csv_file, working_dir, log, geocoding_model)
|
||||
@input_file = input_csv_file
|
||||
@dir = working_dir
|
||||
@log = log
|
||||
@geocoding_model = geocoding_model
|
||||
@non_batch_base_url = config.fetch('non_batch_base_url')
|
||||
@app_id = config.fetch('app_id')
|
||||
@token = config.fetch('token')
|
||||
@mailto = config.fetch('mailto')
|
||||
|
||||
init_rows_count
|
||||
end
|
||||
|
||||
def run
|
||||
init_rows_count
|
||||
@log.append_and_store "Initialized non batch Here geocoding job"
|
||||
@result = File.join(dir, 'generated_csv_out.txt')
|
||||
change_status('running')
|
||||
@total_rows = input_rows
|
||||
@log.append_and_store "Total rows to be processed: #{@total_rows}"
|
||||
::CSV.open(@result, "wb") do |output_csv_file|
|
||||
::CSV.foreach(input_file, headers: true) do |input_row|
|
||||
process_row(input_row, output_csv_file)
|
||||
end
|
||||
end
|
||||
change_status('completed')
|
||||
update_log_stats
|
||||
@log.append_and_store "Non-batch Here geocoding job finished"
|
||||
end
|
||||
|
||||
def used_batch_request?
|
||||
false
|
||||
end
|
||||
|
||||
def cancel; end
|
||||
def update_status; end
|
||||
|
||||
def result
|
||||
@result
|
||||
end
|
||||
|
||||
def request_id
|
||||
# INFO: there's no request_id for non-batch geocodings
|
||||
nil
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def config
|
||||
GeocoderConfig.instance.get
|
||||
end
|
||||
|
||||
def http_client
|
||||
@http_client ||= Carto::Http::Client.get('hires_geocoder',
|
||||
log_requests: true)
|
||||
end
|
||||
|
||||
def input_rows
|
||||
stdout, _stderr, _status = Open3.capture3('wc', '-l', input_file)
|
||||
stdout.to_i
|
||||
rescue
|
||||
0
|
||||
end
|
||||
|
||||
def process_row(input_row, output_csv_file)
|
||||
@processed_rows += 1
|
||||
latitude, longitude = geocode_text(input_row["searchtext"])
|
||||
if !(latitude.nil? || latitude == "") && !(longitude.nil? || longitude == "")
|
||||
@successful_processed_rows += 1
|
||||
output_csv_file.add_row [input_row["searchtext"], 1, 1, latitude, longitude]
|
||||
else
|
||||
@empty_processed_rows += 1
|
||||
end
|
||||
rescue => e
|
||||
@log.append_and_store "Error processing row with search text #{input_row['searchtext']}: #{e.message}"
|
||||
CartoDB.notify_debug("Hires geocoding process row error",
|
||||
error: e.backtrace.join("\n"),
|
||||
searchtext: input_row["searchtext"],
|
||||
backtrace: e.backtrace)
|
||||
@failed_processed_rows += 1
|
||||
end
|
||||
|
||||
def geocode_text(text)
|
||||
options = GEOCODER_OPTIONS.merge(searchtext: text, app_id: app_id, app_code: token)
|
||||
url = "#{non_batch_base_url}?#{URI.encode_www_form(options)}"
|
||||
http_response = http_client.get(url,
|
||||
connecttimeout: HTTP_CONNECTION_TIMEOUT,
|
||||
timeout: HTTP_REQUEST_TIMEOUT)
|
||||
if http_response.success?
|
||||
response = ::JSON.parse(http_response.body)["response"]
|
||||
if response['view'].empty?
|
||||
# no location info for the text input, stop here
|
||||
return [nil, nil]
|
||||
end
|
||||
position = response["view"][0]["result"][0]["location"]["displayPosition"]
|
||||
return position["latitude"], position["longitude"]
|
||||
else
|
||||
CartoDB.notify_debug('Non-batched geocoder failed request', http_response)
|
||||
return [nil, nil]
|
||||
end
|
||||
rescue NoMethodError => e
|
||||
if e.message == %Q(undefined method `[]' for nil:NilClass)
|
||||
CartoDB.notify_debug("Non-batched geocoder couldn't parse response",
|
||||
error: e.backtrace.join("\n"), backtrace: e.backtrace, text: text, response_body: http_response.body)
|
||||
[nil, nil]
|
||||
else
|
||||
raise e
|
||||
end
|
||||
end
|
||||
|
||||
def api_url(arguments, extra_components = nil)
|
||||
arguments.merge!(app_id: app_id, token: token, mailto: mailto)
|
||||
components = [base_url]
|
||||
components << extra_components unless extra_components.nil?
|
||||
components << '?' + URI.encode_www_form(arguments)
|
||||
components.join('/')
|
||||
end
|
||||
|
||||
def init_rows_count
|
||||
@processed_rows = 0
|
||||
@successful_processed_rows = 0
|
||||
@failed_processed_rows = 0
|
||||
@empty_processed_rows = 0
|
||||
end
|
||||
|
||||
def update_log_stats
|
||||
@log.append_and_store "Geocoding non-batch Here job status update. "\
|
||||
"Status: #{@status} --- Processed rows: #{@processed_rows} "\
|
||||
"--- Success: #{@successful_processed_rows} --- Empty: #{@empty_processed_rows} "\
|
||||
"--- Failed: #{@failed_processed_rows}"
|
||||
end
|
||||
|
||||
def change_status(status)
|
||||
@status = status
|
||||
@geocoding_model.state = status
|
||||
@geocoding_model.save
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,53 @@
|
||||
require_relative 'hires_geocoder'
|
||||
require_relative 'hires_batch_geocoder'
|
||||
require_relative 'geocoder_config'
|
||||
|
||||
|
||||
module CartoDB
|
||||
class HiresGeocoderFactory
|
||||
|
||||
BATCH_FILES_OVER = 1100 # Use Here Batch Geocoder API with tables over x rows
|
||||
|
||||
def self.get(input_csv_file, working_dir, log, geocoding_model, number_of_rows = 0)
|
||||
geocoder_class = nil
|
||||
if use_batch_process?(input_csv_file, geocoding_model, number_of_rows)
|
||||
geocoder_class = HiresBatchGeocoder
|
||||
else
|
||||
geocoder_class = HiresGeocoder
|
||||
end
|
||||
|
||||
geocoder_class.new(input_csv_file, working_dir, log, geocoding_model)
|
||||
end
|
||||
|
||||
|
||||
private
|
||||
|
||||
def self.use_batch_process?(input_csv_file, geocoding_model, number_of_rows)
|
||||
# Due we could check this condition to create the geocoder class and we don't
|
||||
# have finished yet the csv file generation, and could be nil, we have to check
|
||||
# multiples conditions. It's sorted by priority
|
||||
if force_batch? || geocoding_model.batched
|
||||
true
|
||||
elsif (not input_csv_file.nil?) && (input_rows(input_csv_file) > BATCH_FILES_OVER)
|
||||
true
|
||||
elsif (not number_of_rows.nil?) && (number_of_rows > BATCH_FILES_OVER)
|
||||
true
|
||||
else
|
||||
false
|
||||
end
|
||||
end
|
||||
|
||||
def self.force_batch?
|
||||
GeocoderConfig.instance.get['force_batch'] || false
|
||||
end
|
||||
|
||||
def self.input_rows(input_csv_file)
|
||||
stdout, _stderr, _status = Open3.capture3('wc', '-l', input_csv_file)
|
||||
stdout.to_i
|
||||
rescue => e
|
||||
CartoDB.notify_exception(e)
|
||||
0
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,11 @@
|
||||
module CartoDB
|
||||
class HiresGeocoderInterface
|
||||
def run
|
||||
raise 'Not implemented'
|
||||
end
|
||||
|
||||
def cancel
|
||||
raise 'Not implemented'
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,2 @@
|
||||
recid,searchtext
|
||||
"Fredericton, Canada","Fredericton, Canada"
|
||||
|
@@ -0,0 +1 @@
|
||||
<?xml version="1.0" encoding="UTF-8" standalone="yes"?><ns2:SearchBatch xmlns:ns2="http://www.navteq.com/lbsp/Search-Batch/1"><Response><MetaInfo><RequestId>0TK8XRhsIAxMi0KNbEu58MkApMVWctFE</RequestId></MetaInfo><Status>cancelled</Status><JobStarted>2013-09-15T18:19:49.000Z</JobStarted><JobFinished>2013-09-15T18:19:54.000Z</JobFinished><TotalCount>3</TotalCount><ValidCount>2</ValidCount><InvalidCount>1</InvalidCount><ProcessedCount>2</ProcessedCount><PendingCount>0</PendingCount><SuccessCount>2</SuccessCount><ErrorCount>0</ErrorCount></Response></ns2:SearchBatch>
|
||||
@@ -0,0 +1 @@
|
||||
<?xml version="1.0" encoding="UTF-8" standalone="yes"?><ns2:SearchBatch xmlns:ns2="http://www.navteq.com/lbsp/Search-Batch/1"><Response><MetaInfo><RequestId>K8DmCWzsZGh4gbawxOuMv2BUcZsIkt7v</RequestId></MetaInfo><Status>submitted</Status><TotalCount>0</TotalCount><ValidCount>0</ValidCount><InvalidCount>0</InvalidCount><ProcessedCount>0</ProcessedCount><PendingCount>0</PendingCount><SuccessCount>0</SuccessCount><ErrorCount>0</ErrorCount></Response></ns2:SearchBatch>
|
||||
@@ -0,0 +1 @@
|
||||
{"response":{"metaInfo":{"timestamp":"2014-02-19T12:29:49.723+0000"},"view":[{"result":[{"relevance":1.0,"matchLevel":"country","matchQuality":{"country":1.0},"location":{"locationId":"AREA_21000001","locationType":"point","displayPosition":{"latitude":38.89037,"longitude":-77.03196},"navigationPosition":[{"latitude":38.89037,"longitude":-77.03196}],"mapView":{"topLeft":{"latitude":49.3845,"longitude":-124.749},"bottomRight":{"latitude":24.5018,"longitude":-66.9406}},"address":{"label":"United States","country":"USA","additionalData":[{"value":"United States","key":"CountryName"}]}}}],"viewId":0}]}}
|
||||
@@ -0,0 +1 @@
|
||||
<?xml version="1.0" encoding="UTF-8" standalone="yes"?><ns2:Error xmlns:ns2="http://www.navteq.com/lbsp/Errors/1" type="ApplicationError" subtype="InvalidInputData"><Details>Input parameter validation failed. JobId: 9rFyj7kbGMmpF50ZUFAkRnroEiOpDOEZ Email Address is missing!</Details><AdditionalData key="mailto"/></ns2:Error>
|
||||
@@ -0,0 +1 @@
|
||||
<?xml version="1.0" encoding="UTF-8" standalone="yes"?><ns2:SearchBatch xmlns:ns2="http://www.navteq.com/lbsp/Search-Batch/1"><Response><MetaInfo><RequestId>0TK8XRhsIAxMi0KNbEu58MkApMVWctFE</RequestId></MetaInfo><Status>completed</Status><JobStarted>2013-09-15T18:19:49.000Z</JobStarted><JobFinished>2013-09-15T18:19:54.000Z</JobFinished><TotalCount>3</TotalCount><ValidCount>2</ValidCount><InvalidCount>1</InvalidCount><ProcessedCount>2</ProcessedCount><PendingCount>0</PendingCount><SuccessCount>2</SuccessCount><ErrorCount>0</ErrorCount></Response></ns2:SearchBatch>
|
||||
@@ -0,0 +1,4 @@
|
||||
recId,searchText,country
|
||||
1,425 W Randolph St, Chicago Illinois 60606,USA
|
||||
2,31 St James Ave Boston MA 02116,USA
|
||||
3,10115 Berlin Invalidenstrasse 117,DEU
|
||||
|
@@ -0,0 +1,4 @@
|
||||
recId,searchText
|
||||
1,425 W Randolph St, Chicago Illinois 60606
|
||||
2,31 St James Ave Boston MA 02116
|
||||
3,10115 Berlin Invalidenstrasse 117
|
||||
|
@@ -0,0 +1,171 @@
|
||||
require_relative '../../../spec/spec_helper'
|
||||
require_relative '../../../spec/rspec_configuration.rb'
|
||||
require_relative '../lib/hires_batch_geocoder'
|
||||
|
||||
# TODO rename to hires_batch_geocoder_spec.rb or split into batch/non-batch
|
||||
|
||||
describe CartoDB::HiresBatchGeocoder do
|
||||
|
||||
before(:each) do
|
||||
@log = mock
|
||||
@log.stubs(:append)
|
||||
@log.stubs(:append_and_store)
|
||||
CartoDB::HiresBatchGeocoder.any_instance.stubs(:config).returns({
|
||||
'base_url' => 'http://wadus.nokia.com',
|
||||
'app_id' => '',
|
||||
'token' => '',
|
||||
'mailto' => ''
|
||||
})
|
||||
@working_dir = Dir.mktmpdir
|
||||
@geocoding_model = FactoryGirl.create(:geocoding, kind: 'high-resolution', formatter: '{street}',
|
||||
remote_id: 'wadus')
|
||||
end
|
||||
|
||||
after(:each) do
|
||||
FileUtils.remove_entry_secure @working_dir
|
||||
end
|
||||
|
||||
describe '#upload' do
|
||||
it 'returns rec_id on success' do
|
||||
stub_api_request 200, 'response_example.xml'
|
||||
filepath = path_to 'without_country.csv'
|
||||
rec_id = CartoDB::HiresBatchGeocoder.new(filepath, @working_dir, @log, @geocoding_model).upload
|
||||
rec_id.should eq "K8DmCWzsZGh4gbawxOuMv2BUcZsIkt7v"
|
||||
end
|
||||
|
||||
it 'raises error on failure' do
|
||||
stub_api_request 400, 'response_failure.xml'
|
||||
filepath = path_to 'without_country.csv'
|
||||
expect {
|
||||
CartoDB::HiresBatchGeocoder.new(filepath, @working_dir, @log, @geocoding_model).upload
|
||||
}.to raise_error('Geocoding API communication failure: Input parameter validation failed. JobId: 9rFyj7kbGMmpF50ZUFAkRnroEiOpDOEZ Email Address is missing!')
|
||||
end
|
||||
end
|
||||
|
||||
describe '#update_status' do
|
||||
before {
|
||||
stub_api_request(200, 'response_status.xml')
|
||||
CartoDB::HiresBatchGeocoder.any_instance.stubs(:request_id).returns('wadus')
|
||||
}
|
||||
let(:geocoder) { CartoDB::HiresBatchGeocoder.new('/tmp/dummy_input_file.csv', @working_dir, @log, @geocoding_model) }
|
||||
|
||||
it "updates status" do
|
||||
expect { geocoder.update_status }.to change(geocoder, :status).from(nil).to('completed')
|
||||
end
|
||||
it "updates processed rows" do
|
||||
expect { geocoder.update_status }.to change(geocoder, :processed_rows).from(nil).to(2)
|
||||
end
|
||||
it "updates total rows" do
|
||||
expect { geocoder.update_status }.to change(geocoder, :total_rows).from(nil).to(3)
|
||||
end
|
||||
end
|
||||
|
||||
describe '#result' do
|
||||
it "saves result file on working directory" do
|
||||
pending 'move to non-batch suite' # TODO
|
||||
filepath = path_to 'without_country.csv'
|
||||
stub_api_request 200, 'response_example_non_batch.json'
|
||||
geocoder = CartoDB::Geocoder.new(default_params.merge(input_file: filepath, force_batch: false))
|
||||
geocoder.upload
|
||||
geocoder.status.should eq 'completed'
|
||||
result_file = geocoder.result
|
||||
File.file?(result_file).should be true
|
||||
File.dirname(result_file).should eq geocoder.dir
|
||||
end
|
||||
end
|
||||
|
||||
describe '#cancel' do
|
||||
before {
|
||||
stub_api_request(200, 'response_cancel.xml')
|
||||
@geocoding_model.remote_id = 'wadus'
|
||||
@geocoding_model.save
|
||||
CartoDB::HiresBatchGeocoder.any_instance.stubs(:request_id).returns('wadus')
|
||||
}
|
||||
let(:geocoder) { CartoDB::HiresBatchGeocoder.new('dummy_input_file.csv', @working_dir, @log, @geocoding_model) }
|
||||
|
||||
it "updates the status" do
|
||||
geocoder.cancel
|
||||
@geocoding_model.state.should eq 'cancelled'
|
||||
end
|
||||
end
|
||||
|
||||
describe '#extract_response_field' do
|
||||
let(:geocoder) { CartoDB::HiresBatchGeocoder.new('dummy_input.csv', @working_dir, @log, @geocoding_model) }
|
||||
let(:response) { File.open(path_to('response_example.xml')).read }
|
||||
|
||||
it 'returns specified element value' do
|
||||
geocoder.send(:extract_response_field, response, '//Response/Status').should == 'submitted'
|
||||
end
|
||||
|
||||
it 'returns nil for missing elements' do
|
||||
CartoDB.expects(:notify_exception).once
|
||||
geocoder.send(:extract_response_field, response, 'MissingField').should == nil
|
||||
end
|
||||
end
|
||||
|
||||
describe '#api_url' do
|
||||
# TODO move to common place for both geocoders
|
||||
before(:each) {
|
||||
CartoDB::HiresBatchGeocoder.any_instance.stubs(:config).returns({
|
||||
'base_url' => '',
|
||||
'app_id' => 'a',
|
||||
'token' => 'b',
|
||||
'mailto' => 'c'
|
||||
})
|
||||
@geocoder = CartoDB::HiresBatchGeocoder.new('dummy_input.csv', @working_dir, @log, @geocoding_model)
|
||||
}
|
||||
|
||||
it 'returns base url by default' do
|
||||
@geocoder.send(:api_url, {}).should == "/wadus/?app_id=a&token=b&mailto=c"
|
||||
end
|
||||
|
||||
it 'allows for api method specification' do
|
||||
@geocoder.send(:api_url, {}, 'all').should == "/wadus/all/?app_id=a&token=b&mailto=c"
|
||||
end
|
||||
|
||||
it 'allows for api attributes specification' do
|
||||
@geocoder.send(:api_url, {attr: 'wadus'}, 'all').should == "/wadus/all/?attr=wadus&app_id=a&token=b&mailto=c"
|
||||
end
|
||||
end
|
||||
|
||||
describe '#geocode_text' do
|
||||
it 'returns lat/lon on success' do
|
||||
pending 'move to non-batched suite' # TODO
|
||||
stub_api_request 200, 'response_example_non_batch.json'
|
||||
g = CartoDB::Geocoder.new(default_params)
|
||||
g.geocode_text("United States").should eq [38.89037, -77.03196]
|
||||
end
|
||||
end
|
||||
|
||||
describe '#used_batch_request?' do
|
||||
it 'returns true if sent a request to hi-res batch api' do
|
||||
pending 'move these to the factory tests' # TODO
|
||||
stub_api_request 200, 'response_example.xml'
|
||||
filepath = path_to 'without_country.csv'
|
||||
geocoder = CartoDB::Geocoder.new(default_params.merge(input_file: filepath))
|
||||
geocoder.used_batch_request?.should eq true
|
||||
end
|
||||
|
||||
it 'returns false if sent the request was non-batched' do
|
||||
pending 'move these to the factory tests' # TODO
|
||||
stub_api_request 200, 'response_example_non_batch.json'
|
||||
filepath = path_to 'without_country.csv'
|
||||
g = CartoDB::Geocoder.new(default_params.merge(force_batch: false, input_file: filepath))
|
||||
g.upload
|
||||
g.used_batch_request?.should eq false
|
||||
end
|
||||
end
|
||||
|
||||
def path_to(filepath)
|
||||
File.expand_path(
|
||||
File.join(File.dirname(__FILE__), "../spec/fixtures/#{filepath}")
|
||||
)
|
||||
end #path_to
|
||||
|
||||
def stub_api_request(code, response_file)
|
||||
response = File.open(path_to(response_file)).read
|
||||
Typhoeus.stub(/.*nokia.com/).and_return(
|
||||
Typhoeus::Response.new(code: code, body: response)
|
||||
)
|
||||
end
|
||||
end # CartoDB::Geocoder
|
||||
@@ -0,0 +1,240 @@
|
||||
require 'tmpdir'
|
||||
require 'fileutils'
|
||||
require_relative '../../../spec/rspec_configuration.rb'
|
||||
require_relative '../../../spec/spec_helper.rb'
|
||||
require_relative '../lib/hires_batch_geocoder'
|
||||
|
||||
|
||||
describe CartoDB::HiresBatchGeocoder do
|
||||
|
||||
RSpec.configure do |config|
|
||||
config.before :each do
|
||||
Typhoeus::Expectation.clear
|
||||
end
|
||||
end
|
||||
|
||||
before(:each) do
|
||||
@log = mock
|
||||
@log.stubs(:append)
|
||||
@log.stubs(:append_and_store)
|
||||
@working_dir = Dir.mktmpdir
|
||||
@input_csv_file = path_to '../../table-geocoder/spec/fixtures/nokia_input.csv'
|
||||
CartoDB::HiresBatchGeocoder.any_instance.stubs(:config).returns({
|
||||
'base_url' => 'batch.example.com',
|
||||
'app_id' => '',
|
||||
'token' => '',
|
||||
'mailto' => ''
|
||||
})
|
||||
@geocoding_model = FactoryGirl.create(:geocoding, kind: 'high-resolution', formatter: '{street}' )
|
||||
@batch_geocoder = CartoDB::HiresBatchGeocoder.new(@input_csv_file, @working_dir, @log, @geocoding_model)
|
||||
end
|
||||
|
||||
after(:each) do
|
||||
FileUtils.rm_f @working_dir
|
||||
end
|
||||
|
||||
describe '#run' do
|
||||
it 'uploads a file to the batch server' do
|
||||
mock_complete_response
|
||||
@batch_geocoder.expects(:upload).once
|
||||
@batch_geocoder.run
|
||||
@geocoding_model.state.should == 'completed'
|
||||
end
|
||||
|
||||
it 'times out if not finished before the DEFAULT_TIMEOUT' do
|
||||
mock_complete_response('running')
|
||||
@batch_geocoder.expects(:upload).once
|
||||
@batch_geocoder.expects(:cancel).once
|
||||
@batch_geocoder.stubs(:default_timeout).returns(-10) # make sure it times out
|
||||
@batch_geocoder.run
|
||||
@geocoding_model.state.should == 'timeout'
|
||||
end
|
||||
end
|
||||
|
||||
describe '#upload' do
|
||||
it 'uploads a file to the batch service' do
|
||||
url = @batch_geocoder.send(:api_url, CartoDB::HiresBatchGeocoder::UPLOAD_OPTIONS)
|
||||
expected_request_id = 'dummy_id'
|
||||
xml_response_body = "<Response><MetaInfo><RequestId>#{expected_request_id}</RequestId></MetaInfo></Response>"
|
||||
response = Typhoeus::Response.new(code: 200, body: xml_response_body)
|
||||
|
||||
Typhoeus.stub(url, method: :post).and_return(response)
|
||||
@batch_geocoder.upload
|
||||
|
||||
@batch_geocoder.request_id.should == expected_request_id
|
||||
@batch_geocoder.used_batch_request?.should == true
|
||||
end
|
||||
|
||||
it 'raises an exception if the api returns != 200' do
|
||||
url = @batch_geocoder.send(:api_url, CartoDB::HiresBatchGeocoder::UPLOAD_OPTIONS)
|
||||
response = Typhoeus::Response.new(code: 401)
|
||||
Typhoeus.stub(url, method: :post).and_return(response)
|
||||
CartoDB.expects(:notify_exception).once
|
||||
|
||||
expect {
|
||||
@batch_geocoder.upload
|
||||
}.to raise_error(RuntimeError, /Geocoding API communication failure/)
|
||||
end
|
||||
|
||||
it 'raises an exception if the api does not return a RequestId' do
|
||||
url = @batch_geocoder.send(:api_url, CartoDB::HiresBatchGeocoder::UPLOAD_OPTIONS)
|
||||
response = Typhoeus::Response.new(code: 200)
|
||||
Typhoeus.stub(url, method: :post).and_return(response)
|
||||
CartoDB.expects(:notify_exception).once
|
||||
|
||||
expect {
|
||||
@batch_geocoder.upload
|
||||
}.to raise_error(RuntimeError, /Could not get the request ID/)
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
describe '#cancel' do
|
||||
it 'sends a cancel put request and gets the status, processed and total rows' do
|
||||
request_id = 'dummy_request_id'
|
||||
@geocoding_model.remote_id = request_id
|
||||
@geocoding_model.save
|
||||
@batch_geocoder.stubs(:request_id).returns(request_id)
|
||||
url = @batch_geocoder.send(:api_url, action: 'cancel')
|
||||
url.should match(%r'/#{request_id}/')
|
||||
url.should match(%r'action=cancel')
|
||||
|
||||
expected_status = 'cancelled'
|
||||
expected_processed_rows = 20
|
||||
expected_success_rows = 17
|
||||
expected_failed_rows = 0
|
||||
expected_empty_rows = 3
|
||||
expected_total_rows = 30
|
||||
|
||||
response_body = <<END_XML
|
||||
<Response>
|
||||
<Status>#{expected_status}</Status>
|
||||
<ProcessedCount>#{expected_processed_rows}</ProcessedCount>
|
||||
<SuccessCount>#{expected_success_rows}</SuccessCount>
|
||||
<ErrorCount>#{expected_failed_rows}</ErrorCount>
|
||||
<InvalidCount>#{expected_empty_rows}</InvalidCount>
|
||||
<TotalCount>#{expected_total_rows}</TotalCount>
|
||||
</Response>
|
||||
END_XML
|
||||
|
||||
response = Typhoeus::Response.new(code: 200, body: response_body)
|
||||
Typhoeus.stub(url, method: :put).and_return(response)
|
||||
@batch_geocoder.cancel
|
||||
@batch_geocoder.status.should == expected_status
|
||||
@batch_geocoder.processed_rows.should == expected_processed_rows
|
||||
@batch_geocoder.total_rows.should == expected_total_rows
|
||||
end
|
||||
end
|
||||
|
||||
describe '#update' do
|
||||
it 'gets the status, processed and total rows by sending a get request' do
|
||||
request_id = 'dummy_request_id'
|
||||
@geocoding_model.remote_id = request_id
|
||||
@geocoding_model.save
|
||||
@batch_geocoder.stubs(:request_id).returns(request_id)
|
||||
url = @batch_geocoder.send(:api_url, action: 'status')
|
||||
url.should match(%r'/#{request_id}/')
|
||||
url.should match(%r'action=status')
|
||||
|
||||
expected_status = 'running'
|
||||
expected_processed_rows = 20
|
||||
expected_success_rows = 17
|
||||
expected_failed_rows = 0
|
||||
expected_empty_rows = 3
|
||||
expected_total_rows = 30
|
||||
|
||||
response_body = <<END_XML
|
||||
<Response>
|
||||
<Status>#{expected_status}</Status>
|
||||
<ProcessedCount>#{expected_processed_rows}</ProcessedCount>
|
||||
<SuccessCount>#{expected_success_rows}</SuccessCount>
|
||||
<ErrorCount>#{expected_failed_rows}</ErrorCount>
|
||||
<InvalidCount>#{expected_empty_rows}</InvalidCount>
|
||||
<TotalCount>#{expected_total_rows}</TotalCount>
|
||||
</Response>
|
||||
END_XML
|
||||
|
||||
response = Typhoeus::Response.new(code: 200, body: response_body)
|
||||
Typhoeus.stub(url, method: :get).and_return(response)
|
||||
@batch_geocoder.update_status
|
||||
@batch_geocoder.status.should == expected_status
|
||||
@batch_geocoder.processed_rows.should == expected_processed_rows
|
||||
@batch_geocoder.total_rows.should == expected_total_rows
|
||||
end
|
||||
end
|
||||
|
||||
describe '#result' do
|
||||
it "raises an exception if there's no request_id from a previous upload" do
|
||||
expect {
|
||||
@batch_geocoder.result
|
||||
}.to raise_error(RuntimeError, /No request_id provided/)
|
||||
end
|
||||
|
||||
it 'downloads the result file from the remote server' do
|
||||
request_id = 'dummy_request_id'
|
||||
@geocoding_model.remote_id = request_id
|
||||
@geocoding_model.save
|
||||
@batch_geocoder.stubs(:request_id).returns(request_id)
|
||||
expected_response_body = 'dummy result file contents'
|
||||
url = @batch_geocoder.send(:api_url, {}, 'result')
|
||||
response = Typhoeus::Response.new(code: 200, body: expected_response_body)
|
||||
Typhoeus.stub(url, method: :get).and_return(response)
|
||||
|
||||
result_file = @batch_geocoder.result
|
||||
File.open(result_file).read.should == expected_response_body
|
||||
|
||||
# it also "memoizes" the result file and avoids further downloads
|
||||
@batch_geocoder.expects(:http_client).never
|
||||
@batch_geocoder.result.should == result_file
|
||||
end
|
||||
|
||||
it 'raises an exception if cannot get a result file' do
|
||||
request_id = 'dummy_request_id'
|
||||
@geocoding_model.remote_id = request_id
|
||||
@geocoding_model.save
|
||||
@batch_geocoder.stubs(:request_id).returns(request_id)
|
||||
expected_response_body = 'dummy result file contents'
|
||||
url = @batch_geocoder.send(:api_url, {}, 'result')
|
||||
response = Typhoeus::Response.new(code: 400)
|
||||
Typhoeus.stub(url, method: :get).and_return(response)
|
||||
|
||||
expect {
|
||||
@batch_geocoder.result
|
||||
}.to raise_error(RuntimeError, /Download request failed/)
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
def path_to(filepath = '')
|
||||
File.expand_path(
|
||||
File.join(File.dirname(__FILE__), "../fixtures/#{filepath}")
|
||||
)
|
||||
end
|
||||
|
||||
def mock_complete_response(state='completed')
|
||||
@geocoding_model.remote_id = 'dummy_id'
|
||||
@geocoding_model.save.reload
|
||||
url = @batch_geocoder.send(:api_url, {action: 'status'})
|
||||
xml_response_body = '<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
|
||||
<ns2:SearchBatch xmlns:ns2="http://www.navteq.com/lbsp/Search-Batch/1">
|
||||
<Response>
|
||||
<MetaInfo>
|
||||
<RequestId>dummy_id</RequestId>
|
||||
</MetaInfo>
|
||||
<Status>'+state+'</Status>
|
||||
<JobStarted>2016-04-08T08:24:05.000Z</JobStarted>
|
||||
<JobFinished>2016-04-08T08:24:39.000Z</JobFinished>
|
||||
<TotalCount>1</TotalCount>
|
||||
<ValidCount>1</ValidCount>
|
||||
<InvalidCount>0</InvalidCount>
|
||||
<ProcessedCount>1</ProcessedCount>
|
||||
<PendingCount>0</PendingCount>
|
||||
<SuccessCount>1</SuccessCount>
|
||||
<ErrorCount>0</ErrorCount>
|
||||
</Response>
|
||||
</ns2:SearchBatch>'
|
||||
response = Typhoeus::Response.new(code: 200, body: xml_response_body)
|
||||
Typhoeus.stub(url, method: :get).and_return(response)
|
||||
end
|
||||
|
||||
end
|
||||
@@ -0,0 +1,68 @@
|
||||
require_relative '../../../spec/rspec_configuration'
|
||||
require_relative '../../../spec/spec_helper'
|
||||
require_relative '../lib/hires_geocoder_factory'
|
||||
require_relative '../lib/geocoder_config'
|
||||
|
||||
describe CartoDB::HiresGeocoderFactory do
|
||||
|
||||
after(:all) do
|
||||
# reset config
|
||||
CartoDB::GeocoderConfig.instance.set(nil)
|
||||
end
|
||||
|
||||
before(:each) do
|
||||
@log = mock
|
||||
@log.stubs(:append)
|
||||
@log.stubs(:append_and_store)
|
||||
@geocoding_model = FactoryGirl.create(:geocoding, kind: 'high-resolution', formatter: '{street}')
|
||||
end
|
||||
|
||||
describe '#get' do
|
||||
it 'returns a HiresGeocoder instance if the input file has less than N rows' do
|
||||
CartoDB::GeocoderConfig.instance.set({
|
||||
'non_batch_base_url' => 'http://api.example.com',
|
||||
'app_id' => 'dummy_app_id',
|
||||
'token' => 'dummy_token',
|
||||
'mailto' => 'dummy_mail_addr'
|
||||
})
|
||||
dummy_input_file = 'dummy_input_file.csv'
|
||||
working_dir = '/tmp/any_dir'
|
||||
input_rows = CartoDB::HiresGeocoderFactory::BATCH_FILES_OVER - 1
|
||||
CartoDB::HiresGeocoderFactory.expects(:input_rows).once.with(dummy_input_file).returns(input_rows)
|
||||
|
||||
CartoDB::HiresGeocoderFactory.get(dummy_input_file, working_dir, @log, @geocoding_model).class.should == CartoDB::HiresGeocoder
|
||||
end
|
||||
|
||||
it 'returns a HiresBatchGeocoder instance if the input file is above N rows' do
|
||||
CartoDB::GeocoderConfig.instance.set({
|
||||
'base_url' => 'http://api.example.com',
|
||||
'app_id' => 'dummy_app_id',
|
||||
'token' => 'dummy_token',
|
||||
'mailto' => 'dummy_mail_addr'
|
||||
})
|
||||
dummy_input_file = 'dummy_input_file.csv'
|
||||
working_dir = '/tmp/any_dir'
|
||||
input_rows = CartoDB::HiresGeocoderFactory::BATCH_FILES_OVER + 1
|
||||
CartoDB::HiresGeocoderFactory.expects(:input_rows).once.with(dummy_input_file).returns(input_rows)
|
||||
|
||||
CartoDB::HiresGeocoderFactory.get(dummy_input_file, working_dir, @log, @geocoding_model).class.should == CartoDB::HiresBatchGeocoder
|
||||
end
|
||||
|
||||
it 'returns a batch geocoder if config has force_batch set to true' do
|
||||
CartoDB::GeocoderConfig.instance.set({
|
||||
'force_batch' => true,
|
||||
'base_url' => 'http://api.example.com',
|
||||
'app_id' => 'dummy_app_id',
|
||||
'token' => 'dummy_token',
|
||||
'mailto' => 'dummy_mail_addr'
|
||||
})
|
||||
dummy_input_file = 'dummy_input_file.csv'
|
||||
working_dir = '/tmp/any_dir'
|
||||
CartoDB::HiresGeocoderFactory.expects(:input_rows).never
|
||||
|
||||
CartoDB::HiresGeocoderFactory.get(dummy_input_file, working_dir, @log, @geocoding_model).class.should == CartoDB::HiresBatchGeocoder
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
end
|
||||
@@ -0,0 +1,135 @@
|
||||
require 'tmpdir'
|
||||
require 'fileutils'
|
||||
require 'csv'
|
||||
require_relative '../../../spec/rspec_configuration'
|
||||
require_relative '../../../spec/spec_helper'
|
||||
require_relative '../lib/hires_geocoder'
|
||||
|
||||
|
||||
describe CartoDB::HiresGeocoder do
|
||||
|
||||
MOCK_COORDINATES = [38.89037, -77.03196]
|
||||
|
||||
RSpec.configure do |config|
|
||||
config.before :each do
|
||||
Typhoeus::Expectation.clear
|
||||
end
|
||||
end
|
||||
|
||||
before(:each) do
|
||||
@working_dir = Dir.mktmpdir
|
||||
@input_csv_file = path_to '../../table-geocoder/spec/fixtures/nokia_input.csv'
|
||||
@log = mock
|
||||
@log.stubs(:append)
|
||||
@log.stubs(:append_and_store)
|
||||
CartoDB::HiresGeocoder.any_instance.stubs(:config).returns({
|
||||
'non_batch_base_url' => 'batch.example.com',
|
||||
'app_id' => '',
|
||||
'token' => '',
|
||||
'mailto' => ''
|
||||
})
|
||||
@geocoding_model = FactoryGirl.create(:geocoding, kind: 'high-resolution', formatter: '{street}')
|
||||
@geocoder = CartoDB::HiresGeocoder.new(@input_csv_file, @working_dir, @log, @geocoding_model)
|
||||
end
|
||||
|
||||
after(:each) do
|
||||
FileUtils.rm_f @working_dir
|
||||
end
|
||||
|
||||
describe '#run' do
|
||||
it 'takes every row from input and calls geocode_text on them' do
|
||||
rows_to_geocode = ::CSV.read(@input_csv_file, headers: true).length
|
||||
@geocoder.expects(:geocode_text).times(rows_to_geocode).returns(MOCK_COORDINATES)
|
||||
@geocoder.run
|
||||
@geocoder.status.should == 'completed'
|
||||
end
|
||||
end
|
||||
|
||||
describe '#process_row' do
|
||||
it 'increments the number of processed rows by one when called' do
|
||||
output_csv_mock = mock
|
||||
output_csv_mock.expects(:add_row).once
|
||||
input_row = {'searchtext' => 'olakase'}
|
||||
@geocoder.expects(:geocode_text).once.returns(MOCK_COORDINATES)
|
||||
@geocoder.processed_rows.should == 0
|
||||
@geocoder.send(:process_row, input_row, output_csv_mock)
|
||||
@geocoder.processed_rows.should == 1
|
||||
end
|
||||
|
||||
it 'adds a row with the expected format when success' do
|
||||
output_csv_mock = mock
|
||||
output_csv_mock.expects(:add_row).once.with ['olakase', 1, 1, MOCK_COORDINATES[0], MOCK_COORDINATES[1]]
|
||||
input_row = {'searchtext' => 'olakase'}
|
||||
@geocoder.expects(:geocode_text).once.returns(MOCK_COORDINATES)
|
||||
@geocoder.send(:process_row, input_row, output_csv_mock)
|
||||
end
|
||||
|
||||
it 'does not add any row when it when geolocation fails' do
|
||||
output_csv_mock = mock
|
||||
output_csv_mock.expects(:add_row).never
|
||||
input_row = {'searchtext' => 'olakase'}
|
||||
@geocoder.expects(:geocode_text).once.returns([nil, nil])
|
||||
@geocoder.send(:process_row, input_row, output_csv_mock)
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
describe '#geocode_text' do
|
||||
it 'sends a request to the non-batched geocoder service and gets a couple of coordinates' do
|
||||
json_response_body = {
|
||||
response: {
|
||||
view: [
|
||||
result: [
|
||||
location: {
|
||||
displayPosition: {
|
||||
latitude: MOCK_COORDINATES[0],
|
||||
longitude: MOCK_COORDINATES[1]
|
||||
}
|
||||
}
|
||||
]
|
||||
]
|
||||
}
|
||||
}.to_json
|
||||
mocked_response = Typhoeus::Response.new(code: 200, body: json_response_body)
|
||||
Typhoeus.stub(//, method: :get).and_return(mocked_response)
|
||||
|
||||
@geocoder.send(:geocode_text, 'Dummy address').should == MOCK_COORDINATES
|
||||
end
|
||||
|
||||
it "returns nil coordinates if the http request doesn't succeed" do
|
||||
mocked_response = Typhoeus::Response.new(code: 500)
|
||||
Typhoeus.stub(//, method: :get).and_return(mocked_response)
|
||||
CartoDB.expects(:notify_debug).with('Non-batched geocoder failed request', mocked_response).once
|
||||
|
||||
@geocoder.send(:geocode_text, 'Dummy address').should == [nil, nil]
|
||||
end
|
||||
|
||||
it 'returns nil coordinates and log a trace if it is not able to parse the response' do
|
||||
input_text = 'Dummy address'
|
||||
json_response_body = {
|
||||
unexpected: 'this response body has unexpected format for whatever reason'
|
||||
}.to_json
|
||||
mocked_response = Typhoeus::Response.new(code: 200, body: json_response_body)
|
||||
Typhoeus.stub(//, method: :get).and_return(mocked_response)
|
||||
CartoDB.expects(:notify_debug).with("Non-batched geocoder couldn't parse response", anything()).once
|
||||
|
||||
@geocoder.send(:geocode_text, input_text).should == [nil, nil]
|
||||
end
|
||||
|
||||
it 'returns nil coordinates and stops there if the response does not contain any location' do
|
||||
input_text = 'Dummy address'
|
||||
json_response_body = '{"response":{"metaInfo":{"timestamp":"2015-07-14T15:33:35.023+0000"},"view":[]}}'
|
||||
mocked_response = Typhoeus::Response.new(code: 200, body: json_response_body)
|
||||
Typhoeus.stub(//, method: :get).and_return(mocked_response)
|
||||
|
||||
@geocoder.send(:geocode_text, input_text).should == [nil, nil]
|
||||
end
|
||||
end
|
||||
|
||||
def path_to(filepath = '')
|
||||
File.expand_path(
|
||||
File.join(File.dirname(__FILE__), "../fixtures/#{filepath}")
|
||||
)
|
||||
end
|
||||
|
||||
end
|
||||
@@ -0,0 +1,21 @@
|
||||
guard 'minitest', test_folders: 'spec' do
|
||||
# with Minitest::Spec
|
||||
|
||||
watch(%r|^spec/unit/(.*)_spec\.rb|)
|
||||
watch(%r|^(.*)\.rb|) { |m| "spec/unit/#{m[1]}_spec.rb" }
|
||||
|
||||
# with Minitest::Unit
|
||||
# watch(%r|^test/(.*)\/?test_(.*)\.rb|)
|
||||
# watch(%r|^lib/(.*)([^/]+)\.rb|) { |m| "test/#{m[1]}test_#{m[2]}.rb" }
|
||||
# watch(%r|^test/test_helper\.rb|) { "test" }
|
||||
|
||||
# Rails 3.2
|
||||
# watch(%r|^app/controllers/(.*)\.rb|) { |m| "test/controllers/#{m[1]}_test.rb" }
|
||||
# watch(%r|^app/helpers/(.*)\.rb|) { |m| "test/helpers/#{m[1]}_test.rb" }
|
||||
# watch(%r|^app/models/(.*)\.rb|) { |m| "test/unit/#{m[1]}_test.rb" }
|
||||
|
||||
# Rails
|
||||
# watch(%r|^app/controllers/(.*)\.rb|) { |m| "test/functional/#{m[1]}_test.rb" }
|
||||
# watch(%r|^app/helpers/(.*)\.rb|) { |m| "test/helpers/#{m[1]}_test.rb" }
|
||||
# watch(%r|^app/models/(.*)\.rb|) { |m| "test/unit/#{m[1]}_test.rb" }
|
||||
end
|
||||
@@ -0,0 +1,17 @@
|
||||
require 'rake/testtask'
|
||||
|
||||
Rake::TestTask.new do |t|
|
||||
t.libs << "test"
|
||||
t.pattern = "spec/**/*_spec.rb"
|
||||
end
|
||||
|
||||
Rake::TestTask.new('test:unit') do |t|
|
||||
t.libs << "test"
|
||||
t.pattern = "spec/unit/**/*_spec.rb"
|
||||
end
|
||||
|
||||
Rake::TestTask.new('test:acceptance') do |t|
|
||||
t.libs << "test"
|
||||
t.pattern = "spec/acceptance/**/*_spec.rb"
|
||||
end
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
module CartoDB
|
||||
module Importer2
|
||||
module QuotaCheckHelpers
|
||||
def raise_if_over_storage_quota(requested_quota: 0, available_quota: 0, user_id: nil)
|
||||
quota_overage = requested_quota - available_quota
|
||||
|
||||
if quota_overage > 0
|
||||
report_over_quota(user_id, quota_overage: quota_overage) if user_id
|
||||
|
||||
raise StorageQuotaExceededError.new
|
||||
end
|
||||
end
|
||||
|
||||
def report_over_quota(user_id, quota_overage: 0)
|
||||
Carto::Tracking::Events::ExceededQuota.new(user_id,
|
||||
user_id: user_id,
|
||||
quota_overage: quota_overage).report
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,10 @@
|
||||
require_relative './importer/column'
|
||||
require_relative './importer/downloader'
|
||||
require_relative './importer/datasource_downloader'
|
||||
require_relative './importer/georeferencer'
|
||||
require_relative './importer/job'
|
||||
require_relative './importer/loader'
|
||||
require_relative './importer/ogr2ogr'
|
||||
require_relative './importer/runner'
|
||||
require_relative './importer/source_file'
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
module CartoDB
|
||||
module Importer2
|
||||
|
||||
class CartodbfyTime
|
||||
|
||||
@@instances = {}
|
||||
|
||||
# Gets an instance unique per process + data_import_id
|
||||
def self.instance(data_import_id)
|
||||
@@instances[data_import_id] ||= new
|
||||
end
|
||||
|
||||
def initialize
|
||||
@cartodbfy_time = 0.0
|
||||
end
|
||||
|
||||
def add(elapsed_time)
|
||||
@cartodbfy_time += elapsed_time
|
||||
end
|
||||
|
||||
def get
|
||||
return @cartodbfy_time
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,291 @@
|
||||
require 'active_support/time'
|
||||
|
||||
require_relative './job'
|
||||
require_relative './string_sanitizer'
|
||||
require_relative './exceptions'
|
||||
require_relative './query_batcher'
|
||||
|
||||
module CartoDB
|
||||
module Importer2
|
||||
class Column
|
||||
DEFAULT_SRID = 4326
|
||||
WKB_RE = /^\d{2}/
|
||||
GEOJSON_RE = /{.*(type|coordinates).*(type|coordinates).*}/
|
||||
WKT_RE = /POINT|LINESTRING|POLYGON/
|
||||
KML_MULTI_RE = /<Line|<Polygon/
|
||||
KML_POINT_RE = /<Point>/
|
||||
DEFAULT_SCHEMA = 'cdb_importer'
|
||||
DIRECT_STATEMENT_TIMEOUT = 1.hour * 1000
|
||||
# @see config/initializers/carto_db.rb -> POSTGRESQL_RESERVED_WORDS
|
||||
RESERVED_WORDS = %w{ ALL ANALYSE ANALYZE AND ANY ARRAY AS ASC ASYMMETRIC
|
||||
AUTHORIZATION BETWEEN BINARY BOTH CASE CAST CHECK
|
||||
COLLATE COLUMN CONSTRAINT CREATE CROSS CURRENT_DATE
|
||||
CURRENT_ROLE CURRENT_TIME CURRENT_TIMESTAMP
|
||||
CURRENT_USER DEFAULT DEFERRABLE DESC DISTINCT DO
|
||||
ELSE END EXCEPT FALSE FOR FOREIGN FREEZE FROM FULL
|
||||
GRANT GROUP HAVING ILIKE IN INITIALLY INNER INTERSECT
|
||||
INTO IS ISNULL JOIN LEADING LEFT LIKE LIMIT LOCALTIME
|
||||
LOCALTIMESTAMP NATURAL NEW NOT NOTNULL NULL OFF
|
||||
OFFSET OLD ON ONLY OR ORDER OUTER OVERLAPS PLACING
|
||||
PRIMARY REFERENCES RIGHT SELECT SESSION_USER SIMILAR
|
||||
SOME SYMMETRIC TABLE THEN TO TRAILING TRUE UNION
|
||||
UNIQUE USER USING VERBOSE WHEN WHERE XMIN XMAX
|
||||
FORMAT CONTROLLER ACTION
|
||||
}
|
||||
|
||||
def initialize(db, table_name, column_name, user, schema = DEFAULT_SCHEMA, job = nil, logger = nil, capture_exceptions = true)
|
||||
@job = job || Job.new({logger: logger})
|
||||
@db = db
|
||||
@table_name = table_name
|
||||
@column_name = column_name.to_sym
|
||||
@schema = schema
|
||||
@capture_exceptions = capture_exceptions
|
||||
@user = user
|
||||
|
||||
@from_geojson_with_transform = false
|
||||
end
|
||||
|
||||
def mark_as_from_geojson_with_transform
|
||||
@from_geojson_with_transform = true
|
||||
end
|
||||
|
||||
def type
|
||||
db.schema(table_name, reload: true, schema: schema)
|
||||
.select { |column_details|
|
||||
column_details.first == column_name
|
||||
}.last.last.fetch(:db_type)
|
||||
end
|
||||
|
||||
def geometrify
|
||||
job.log 'geometrifying'
|
||||
raise "empty column #{column_name}" if empty?
|
||||
convert_from_wkt if wkt?
|
||||
convert_from_kml_multi if kml_multi?
|
||||
convert_from_kml_point if kml_point?
|
||||
convert_from_geojson_with_transform if geojson? && @from_geojson_with_transform
|
||||
convert_from_geojson if geojson?
|
||||
|
||||
cast_to('geometry')
|
||||
convert_to_2d
|
||||
job.log 'geometrified'
|
||||
self
|
||||
end
|
||||
|
||||
def convert_from_wkt
|
||||
#TODO: @capture_exceptions
|
||||
job.log 'Converting geometry from WKT to WKB'
|
||||
@user.db_service.in_database_direct_connection(statement_timeout: DIRECT_STATEMENT_TIMEOUT) do |user_direct_conn|
|
||||
user_direct_conn.run(%Q{
|
||||
UPDATE #{qualified_table_name}
|
||||
SET #{column_name} = ST_GeomFromText(#{column_name}, #{DEFAULT_SRID})
|
||||
})
|
||||
end
|
||||
self
|
||||
end
|
||||
|
||||
def convert_from_geojson_with_transform
|
||||
# 1) cast to proper null the geom column
|
||||
db.run(%Q{
|
||||
UPDATE #{qualified_table_name}
|
||||
SET #{column_name} = NULL
|
||||
WHERE #{column_name} = ''
|
||||
})
|
||||
|
||||
# 2) Normal geojson behavior
|
||||
#TODO: @capture_exceptions
|
||||
job.log 'Converting geometry from GeoJSON with transform to WKB'
|
||||
@user.db_service.in_database_direct_connection(statement_timeout: DIRECT_STATEMENT_TIMEOUT) do |user_direct_conn|
|
||||
user_direct_conn.run(%Q{
|
||||
UPDATE #{qualified_table_name}
|
||||
SET #{column_name} = public.ST_SetSRID(public.ST_GeomFromGeoJSON(#{column_name}), #{DEFAULT_SRID})
|
||||
})
|
||||
end
|
||||
|
||||
self
|
||||
end
|
||||
|
||||
def convert_from_geojson
|
||||
#TODO: @capture_exceptions
|
||||
job.log 'Converting geometry from GeoJSON to WKB'
|
||||
@user.db_service.in_database_direct_connection(statement_timeout: DIRECT_STATEMENT_TIMEOUT) do |user_direct_conn|
|
||||
user_direct_conn.run(%Q{
|
||||
UPDATE #{qualified_table_name}
|
||||
SET #{column_name} = public.ST_SetSRID(public.ST_GeomFromGeoJSON(#{column_name}), #{DEFAULT_SRID})
|
||||
})
|
||||
end
|
||||
|
||||
self
|
||||
end
|
||||
|
||||
def convert_from_kml_point
|
||||
#TODO: @capture_exceptions
|
||||
job.log 'Converting geometry from KML point to WKB'
|
||||
@user.db_service.in_database_direct_connection(statement_timeout: DIRECT_STATEMENT_TIMEOUT) do |user_direct_conn|
|
||||
user_direct_conn.run(%Q{
|
||||
UPDATE #{qualified_table_name}
|
||||
SET #{column_name} = public.ST_SetSRID(public.ST_GeomFromKML(#{column_name}),#{DEFAULT_SRID})
|
||||
})
|
||||
end
|
||||
end
|
||||
|
||||
def convert_from_kml_multi
|
||||
#TODO: @capture_exceptions
|
||||
job.log 'Converting geometry from KML multi to WKB'
|
||||
@user.db_service.in_database_direct_connection(statement_timeout: DIRECT_STATEMENT_TIMEOUT) do |user_direct_conn|
|
||||
user_direct_conn.run(%Q{
|
||||
UPDATE #{qualified_table_name}
|
||||
SET #{column_name} = public.ST_SetSRID(public.ST_Multi(public.ST_GeomFromKML(#{column_name})),#{DEFAULT_SRID})
|
||||
})
|
||||
end
|
||||
end
|
||||
|
||||
def convert_to_2d
|
||||
#TODO: @capture_exceptions
|
||||
job.log 'Converting to 2D point'
|
||||
|
||||
@user.db_service.in_database_direct_connection(statement_timeout: DIRECT_STATEMENT_TIMEOUT) do |user_direct_conn|
|
||||
user_direct_conn.run(%Q{
|
||||
UPDATE #{qualified_table_name}
|
||||
SET #{column_name} = public.ST_Force2D(#{column_name})
|
||||
})
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
def wkb?
|
||||
!!(sample.to_s =~ WKB_RE)
|
||||
end
|
||||
|
||||
def wkt?
|
||||
!!(sample.to_s =~ WKT_RE)
|
||||
end
|
||||
|
||||
def geojson?
|
||||
!!(sample.to_s =~ GEOJSON_RE)
|
||||
end
|
||||
|
||||
def kml_point?
|
||||
!!(sample.to_s =~ KML_POINT_RE)
|
||||
end
|
||||
|
||||
def kml_multi?
|
||||
!!(sample.to_s =~ KML_MULTI_RE)
|
||||
end
|
||||
|
||||
def cast_to(type)
|
||||
job.log "casting #{column_name} to #{type}"
|
||||
|
||||
@user.db_service.in_database_direct_connection(statement_timeout: DIRECT_STATEMENT_TIMEOUT) do |user_direct_conn|
|
||||
user_direct_conn.run(%Q{
|
||||
ALTER TABLE #{qualified_table_name}
|
||||
ALTER #{column_name}
|
||||
TYPE #{type}
|
||||
USING #{column_name}::#{type}
|
||||
})
|
||||
end
|
||||
self
|
||||
end
|
||||
|
||||
def sample
|
||||
return nil if empty?
|
||||
records_with_data.first.fetch(column_name)
|
||||
end
|
||||
|
||||
def empty?
|
||||
records_with_data.empty?
|
||||
end
|
||||
|
||||
def records_with_data
|
||||
@records_with_data ||= db[%Q{
|
||||
SELECT #{column_name} FROM "#{schema}"."#{table_name}"
|
||||
WHERE #{column_name} IS NOT NULL
|
||||
AND #{column_name}::text != ''
|
||||
LIMIT 1
|
||||
}]
|
||||
end
|
||||
|
||||
def rename_to(new_name)
|
||||
return self if new_name.to_s == column_name.to_s
|
||||
|
||||
job.log "Renaming column #{column_name} TO #{new_name}"
|
||||
|
||||
db.run(%Q{
|
||||
ALTER TABLE "#{schema}"."#{table_name}"
|
||||
RENAME COLUMN "#{column_name}" TO "#{new_name}"
|
||||
})
|
||||
@column_name = new_name
|
||||
end
|
||||
|
||||
def geometry_type
|
||||
sample = db[%Q{
|
||||
SELECT public.GeometryType(ST_Force2D(#{column_name}::geometry))
|
||||
AS type
|
||||
FROM #{schema}.#{table_name}
|
||||
WHERE #{column_name} IS NOT NULL
|
||||
LIMIT 1
|
||||
}].first
|
||||
sample && sample.fetch(:type)
|
||||
end
|
||||
|
||||
def drop
|
||||
db.run(%Q{
|
||||
ALTER TABLE #{qualified_table_name}
|
||||
DROP COLUMN IF EXISTS #{column_name}
|
||||
})
|
||||
end
|
||||
|
||||
# Replace empty strings by nulls to avoid cast errors
|
||||
def empty_lines_to_nulls
|
||||
# first timeout crash
|
||||
job.log 'replace empty strings by nulls?'
|
||||
column_id = column_name.to_sym
|
||||
column_type = nil
|
||||
db.schema(table_name).each do |colid, coldef|
|
||||
if colid == column_id
|
||||
column_type = coldef[:type]
|
||||
end
|
||||
end
|
||||
if column_type != nil && column_type == :string
|
||||
#TODO: @capture_exceptions
|
||||
job.log 'string column found, replacing'
|
||||
@user.db_service.in_database_direct_connection(statement_timeout: DIRECT_STATEMENT_TIMEOUT) do |user_direct_conn|
|
||||
user_direct_conn.run(%Q{
|
||||
UPDATE #{qualified_table_name}
|
||||
SET #{column_name}=NULL
|
||||
WHERE #{column_name}=''
|
||||
})
|
||||
end
|
||||
else
|
||||
job.log 'no string column found, nothing replaced'
|
||||
end
|
||||
end
|
||||
|
||||
def sanitize
|
||||
rename_to(sanitized_name)
|
||||
end
|
||||
|
||||
def sanitized_name
|
||||
name = StringSanitizer.new.sanitize(column_name.to_s)
|
||||
return name unless reserved?(name) || unsupported?(name)
|
||||
"_#{name}"
|
||||
end
|
||||
|
||||
def reserved?(name)
|
||||
RESERVED_WORDS.include?(name.upcase)
|
||||
end
|
||||
|
||||
def unsupported?(name)
|
||||
name !~ /^[a-zA-Z_]/
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
attr_reader :job, :db, :table_name, :column_name, :schema
|
||||
|
||||
def qualified_table_name
|
||||
%Q("#{schema}"."#{table_name}")
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
@@ -0,0 +1,156 @@
|
||||
require 'carto/connector'
|
||||
require_relative 'exceptions'
|
||||
require_relative 'georeferencer'
|
||||
require_relative 'runner_helper'
|
||||
|
||||
module CartoDB
|
||||
module Importer2
|
||||
# ConnectorRunner runs connector-based imports to import datasets from external databases.
|
||||
#
|
||||
# ConnectorRunner has the same public API as Runner, but
|
||||
# instead of downloading and unpacking files and then processing them with ogr2ogr to import them into
|
||||
# a user database table, a Connector instance connects directly to a remote database and imports
|
||||
# the specified dataset into a table in the user's database. Currently FDW are used for this.
|
||||
#
|
||||
# A connector runner is defined by a hash of parameters; the `provider` parameter is always required and specifies
|
||||
# the Connector class, which in turn defines the rest of valid parameters for the connector.
|
||||
#
|
||||
class ConnectorRunner
|
||||
include CartoDB::Importer2::RunnerHelper
|
||||
|
||||
attr_reader :results, :log, :job, :warnings
|
||||
attr_accessor :stats
|
||||
|
||||
def initialize(connector_source, options = {})
|
||||
@pg_options = options[:pg]
|
||||
@log = options[:log] || new_logger
|
||||
@job = options[:job] || new_job(@log, @pg_options)
|
||||
@user = options[:user]
|
||||
@collision_strategy = options[:collision_strategy]
|
||||
@georeferencer = options[:georeferencer] || new_georeferencer(@job)
|
||||
|
||||
@id = @job.id
|
||||
@unique_suffix = @id.delete('-')
|
||||
@json_params = JSON.parse(connector_source)
|
||||
extract_params
|
||||
@connector = Carto::Connector.new(@params, user: @user, logger: @log)
|
||||
@results = []
|
||||
@tracker = nil
|
||||
@stats = {}
|
||||
@warnings = {}
|
||||
@connector.check_availability!
|
||||
end
|
||||
|
||||
def run(tracker = nil)
|
||||
@tracker = tracker
|
||||
@job.log "ConnectorRunner #{@json_params.except('connection').to_json}"
|
||||
# TODO: logging with CartoDB::Logger
|
||||
table_name = @job.table_name
|
||||
if should_import?(@connector.remote_table_name)
|
||||
@job.log "Copy connected table"
|
||||
warnings = @connector.copy_table(schema_name: @job.schema, table_name: @job.table_name)
|
||||
@job.log 'Georeference geometry column'
|
||||
georeference
|
||||
@warnings.merge! warnings if warnings.present?
|
||||
else
|
||||
@job.log "Table #{table_name} won't be imported"
|
||||
end
|
||||
rescue => error
|
||||
@job.log "ConnectorRunner Error #{error}"
|
||||
@results.push result_for(@job.schema, table_name, error)
|
||||
else
|
||||
if should_import?(@connector.remote_table_name)
|
||||
@job.log "ConnectorRunner created table #{table_name}"
|
||||
@job.log "job schema: #{@job.schema}"
|
||||
@results.push result_for(@job.schema, table_name)
|
||||
end
|
||||
end
|
||||
|
||||
def georeference
|
||||
@georeferencer.run
|
||||
rescue => error
|
||||
@job.log "ConnectorRunner Error while georeference #{error}"
|
||||
end
|
||||
|
||||
def remote_data_updated?
|
||||
@connector.remote_data_updated?
|
||||
end
|
||||
|
||||
def tracker
|
||||
@tracker || lambda { |state| state }
|
||||
end
|
||||
|
||||
def visualizations
|
||||
# This method is needed to make the interface of ConnectorRunner compatible with Runner
|
||||
[]
|
||||
end
|
||||
|
||||
# General availability of connectors for a user
|
||||
def self.check_availability!(user)
|
||||
Carto::Connector.check_availability! user
|
||||
end
|
||||
|
||||
def etag
|
||||
# This method is needed to make the interface of ConnectorRunner compatible with Runner,
|
||||
# but we have no meaningful data to return here.
|
||||
end
|
||||
|
||||
def checksum
|
||||
# This method is needed to make the interface of ConnectorRunner compatible with Runner,
|
||||
# but we have no meaningful data to return here.
|
||||
end
|
||||
|
||||
def last_modified
|
||||
# This method is needed to make the interface of ConnectorRunner compatible with Runner,
|
||||
# but we have no meaningful data to return here.
|
||||
end
|
||||
|
||||
def provider_name
|
||||
@connector.provider_name
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
# Parse @json_params and extract @params
|
||||
def extract_params
|
||||
@params = Carto::Connector::Parameters.new(@json_params)
|
||||
end
|
||||
|
||||
def result_table_name
|
||||
Carto::DB::Sanitize.sanitize_identifier @connector.remote_table_name
|
||||
end
|
||||
|
||||
def result_for(schema, table_name, error = nil)
|
||||
@job.success_status = !error
|
||||
@job.logger.store
|
||||
Result.new(
|
||||
name: result_table_name,
|
||||
schema: schema,
|
||||
tables: [table_name],
|
||||
success: @job.success_status,
|
||||
error_code: error_for(error),
|
||||
log_trace: @job.logger.to_s,
|
||||
support_tables: []
|
||||
)
|
||||
end
|
||||
|
||||
def new_logger
|
||||
CartoDB::Log.new(type: CartoDB::Log::TYPE_DATA_IMPORT)
|
||||
end
|
||||
|
||||
def new_job(log, pg_options)
|
||||
Job.new(logger: log, pg_options: pg_options)
|
||||
end
|
||||
|
||||
def new_georeferencer(job)
|
||||
Georeferencer.new(job.db, job.table_name, {}, Georeferencer::DEFAULT_SCHEMA, job)
|
||||
end
|
||||
|
||||
UNKNOWN_ERROR_CODE = 99999
|
||||
|
||||
def error_for(exception)
|
||||
exception && ERRORS_MAP.fetch(exception.class, UNKNOWN_ERROR_CODE)
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,223 @@
|
||||
require_relative 'ip_checker'
|
||||
require_relative 'table_sampler'
|
||||
require_relative 'namedplaces_guesser'
|
||||
|
||||
require_relative '../../../../lib/cartodb/stats/importer'
|
||||
|
||||
module CartoDB
|
||||
module Importer2
|
||||
class ContentGuesser
|
||||
|
||||
SQLAPI_CALLS_TIMEOUT = 45
|
||||
COUNTRIES_COLUMN = 'name_'
|
||||
COUNTRIES_QUERY = "SELECT #{COUNTRIES_COLUMN} FROM admin0_synonyms"
|
||||
DEFAULT_MINIMUM_ENTROPY = 0.9
|
||||
ID_COLUMNS = ['ogc_fid', 'gid', 'cartodb_id', 'objectid'].freeze
|
||||
|
||||
attr_reader :country_name_normalizer
|
||||
|
||||
def initialize(db, table_name, schema, options, job=nil)
|
||||
@db = db
|
||||
@table_name = table_name
|
||||
@schema = schema
|
||||
@options = options
|
||||
@job = job
|
||||
@importer_stats = CartoDB::Stats::Importer.instance
|
||||
@country_name_normalizer = Proc.new {|str| str.nil? ? '' : str.gsub(/[^a-zA-Z\u00C0-\u00ff]+/, '').downcase }
|
||||
end
|
||||
|
||||
def set_importer_stats(importer_stats)
|
||||
@importer_stats = importer_stats
|
||||
end
|
||||
|
||||
def enabled?
|
||||
@options[:guessing][:enabled] rescue false
|
||||
end
|
||||
|
||||
def country_column
|
||||
return nil if not enabled?
|
||||
columns.each do |column|
|
||||
return column[:column_name] if is_country_column? column
|
||||
end
|
||||
nil
|
||||
end
|
||||
|
||||
def namedplaces
|
||||
@namedplaces ||= NamedplacesGuesser.new(self)
|
||||
end
|
||||
|
||||
def ip_column
|
||||
return nil if not enabled?
|
||||
columns.each do |column|
|
||||
return column[:column_name] if is_ip_column? column
|
||||
end
|
||||
nil
|
||||
end
|
||||
|
||||
def columns
|
||||
@columns ||= @db[%Q(
|
||||
SELECT column_name, data_type
|
||||
FROM information_schema.columns
|
||||
WHERE table_name = '#{@table_name}' AND table_schema = '#{@schema}'
|
||||
)]
|
||||
end
|
||||
|
||||
def is_country_column?(column)
|
||||
return false unless is_text_type? column
|
||||
entropy = metric_entropy(column, country_name_normalizer)
|
||||
if entropy < minimum_entropy
|
||||
false
|
||||
else
|
||||
proportion = country_proportion(column)
|
||||
if proportion < threshold
|
||||
false
|
||||
else
|
||||
log_country_guessing_match_metrics(proportion)
|
||||
true
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
def log_country_guessing_match_metrics(proportion)
|
||||
@importer_stats.gauge('country_proportion', proportion)
|
||||
end
|
||||
|
||||
def log_ip_guessing_match_metrics(proportion)
|
||||
@importer_stats.gauge('ip_proportion', proportion)
|
||||
end
|
||||
|
||||
def is_ip_column?(column)
|
||||
return false unless is_text_type? column
|
||||
proportion = ip_proportion(column)
|
||||
if proportion > threshold
|
||||
log "ip_proportion(#{column[:column_name]}) = #{proportion}; threshold = #{threshold}; sample.count = #{sample.count}"
|
||||
log "sample.first(4) = #{sample.first(4)}"
|
||||
log_ip_guessing_match_metrics(proportion)
|
||||
true
|
||||
else
|
||||
false
|
||||
end
|
||||
end
|
||||
|
||||
|
||||
# See http://en.wikipedia.org/wiki/Entropy_(information_theory)
|
||||
# See http://www.shannonentropy.netmark.pl/
|
||||
#
|
||||
# Returns 0.0 if all elements in the column are repeated
|
||||
# Returns 1.0 if all elements in the column are different
|
||||
def metric_entropy(column, normalizer=nil)
|
||||
shannon_entropy(column, normalizer) / Math.log(sample.count)
|
||||
end
|
||||
|
||||
def shannon_entropy(column, normalizer)
|
||||
sum = 0.0
|
||||
frequencies(column, normalizer).each { |freq| sum += (freq * Math.log(freq)) }
|
||||
return sum.abs
|
||||
end
|
||||
|
||||
# Returns an array with the relative frequencies of the elements of that column
|
||||
def frequencies(column, normalizer)
|
||||
frequency_table = {}
|
||||
column_name_sym = column[:column_name].to_sym
|
||||
if normalizer
|
||||
sample.each do |row|
|
||||
elem = normalizer.call(row[column_name_sym])
|
||||
update_frequency_element(frequency_table, elem)
|
||||
end
|
||||
else
|
||||
sample.each do |row|
|
||||
elem = row[column_name_sym]
|
||||
update_frequency_element(frequency_table, elem)
|
||||
end
|
||||
end
|
||||
length = sample.count.to_f
|
||||
frequency_table.map { |key, value| value / length }
|
||||
end
|
||||
|
||||
def country_proportion(column)
|
||||
column_name_sym = column[:column_name].to_sym
|
||||
matches = sample.count { |row| countries.include? country_name_normalizer.call(row[column_name_sym]) }
|
||||
country_proportion = matches.to_f / sample.count
|
||||
log "country_proportion(#{column[:column_name]}) = #{country_proportion}"
|
||||
country_proportion
|
||||
end
|
||||
|
||||
def log(msg)
|
||||
@job.log msg if @job
|
||||
end
|
||||
|
||||
def sample
|
||||
@sample ||= TableSampler.new(@db, qualified_table_name, id_column, sample_size).sample
|
||||
end
|
||||
|
||||
def id_column
|
||||
return @id_column if @id_column
|
||||
columns.each do |column|
|
||||
if ID_COLUMNS.include? column[:column_name]
|
||||
@id_column = column[:column_name]
|
||||
return @id_column
|
||||
end
|
||||
end
|
||||
raise ContentGuesserException, "Couldn't find an id column for table #{qualified_table_name}"
|
||||
end
|
||||
|
||||
def ip_proportion(column)
|
||||
column_name_sym = column[:column_name].to_sym
|
||||
matches = sample.count { |row| IpChecker.is_ip?(row[column_name_sym]) }
|
||||
matches.to_f / sample.count
|
||||
end
|
||||
|
||||
def threshold
|
||||
@options[:guessing][:threshold]
|
||||
end
|
||||
|
||||
def is_text_type? column
|
||||
['character varying', 'varchar', 'text'].include? column[:data_type]
|
||||
end
|
||||
|
||||
def sample_size
|
||||
@options[:guessing][:sample_size]
|
||||
end
|
||||
|
||||
def minimum_entropy
|
||||
@minimum_entropy ||= @options[:guessing].fetch(:minimum_entropy, DEFAULT_MINIMUM_ENTROPY)
|
||||
end
|
||||
|
||||
def countries
|
||||
return @countries if @countries
|
||||
@countries = Set.new()
|
||||
geocoder_sql_api.fetch(COUNTRIES_QUERY).each do |country|
|
||||
country_name = country[COUNTRIES_COLUMN]
|
||||
@countries.add country_name if country_name.length >= 2
|
||||
end
|
||||
@countries
|
||||
end
|
||||
|
||||
def geocoder_sql_api
|
||||
@geocoder_sql_api ||= CartoDB::SQLApi.new(
|
||||
@options[:geocoder][:internal].merge({ timeout: SQLAPI_CALLS_TIMEOUT })
|
||||
)
|
||||
end
|
||||
|
||||
attr_writer :geocoder_sql_api
|
||||
|
||||
def qualified_table_name
|
||||
%Q("#{@schema}"."#{@table_name}")
|
||||
end
|
||||
|
||||
private
|
||||
|
||||
def update_frequency_element(frequency_table, elem)
|
||||
if frequency_table.key?(elem)
|
||||
frequency_table[elem] += 1
|
||||
else
|
||||
frequency_table[elem] = 1
|
||||
end
|
||||
end
|
||||
|
||||
end
|
||||
|
||||
class ContentGuesserException < StandardError; end
|
||||
|
||||
end
|
||||
end
|
||||
@@ -0,0 +1,272 @@
|
||||
require 'csv'
|
||||
require 'charlock_holmes'
|
||||
require 'tempfile'
|
||||
require 'fileutils'
|
||||
require_relative './job'
|
||||
require_relative './source_file'
|
||||
require_relative './unp'
|
||||
|
||||
module CartoDB
|
||||
module Importer2
|
||||
class CsvNormalizer
|
||||
|
||||
LINE_SIZE_FOR_CLEANING = 5000
|
||||
LINES_FOR_DETECTION = 100 # How many lines to read?
|
||||
SAMPLE_READ_LIMIT = 500000 # Read big enough sample bytes for the encoding sampling
|
||||
COMMON_DELIMITERS = [',', "\t", ' ', ';', '|'].freeze
|
||||
DELIMITER_WEIGHTS = { ',' => 2, "\t" => 2, ' ' => 1, ';' => 2, '|' => 2 }.freeze
|
||||
DEFAULT_DELIMITER = ','
|
||||
DEFAULT_ENCODING = 'UTF-8'
|
||||
DEFAULT_QUOTE = '"'
|
||||
OUTPUT_DELIMITER = ',' # Normalized CSVs will use this delimiter
|
||||
ENCODING_CONFIDENCE = 28
|
||||
ACCEPTABLE_ENCODINGS = %w{ ISO-8859-1 ISO-8859-2 UTF-8 }
|
||||
REVERSE_LINE_FEED = "\x8D"
|
||||
|
||||
|
||||
def initialize(filepath, job = nil, importer_config = nil)
|
||||
@filepath = filepath
|
||||
@job = job || Job.new
|
||||
@delimiter = nil
|
||||
@force_normalize = false
|
||||
@encoding = nil
|
||||
@importer_config = importer_config
|
||||
end
|
||||
|
||||
def force_normalize
|
||||
@force_normalize = true
|
||||
end
|
||||
|
||||
# @throws MalformedCSVException
|
||||
def run
|
||||
return self unless File.exists?(filepath)
|
||||
|
||||
detect_delimiter
|
||||
|
||||
begin
|
||||
return self unless (needs_normalization? || @force_normalize)
|
||||
rescue CSV::MalformedCSVError => ex
|
||||
raise MalformedCSVException.new(ex.message)
|
||||
end
|
||||
|
||||
normalize(temporary_filepath)
|
||||
release
|
||||
File.rename(temporary_filepath, filepath)
|
||||
FileUtils.rm_rf(temporary_directory)
|
||||
self.temporary_directory = nil
|
||||
self
|
||||
end
|
||||
|
||||
def detect_delimiter
|
||||
|
||||
# Calculate variances of the N first lines for each delimiter, then grab the one that changes less
|
||||
@delimiter = DEFAULT_DELIMITER unless first_line
|
||||
|
||||
lines_for_detection = Array.new
|
||||
|
||||
LINES_FOR_DETECTION.times {
|
||||
line = stream.gets
|
||||
lines_for_detection << remove_quoted_strings(line) unless line.nil?
|
||||
}
|
||||
|
||||
stream.rewind
|
||||
|
||||
# Maybe gets was not able to discern line breaks, try manually:
|
||||
if lines_for_detection.size == 1
|
||||
lines_for_detection = lines_for_detection.first
|
||||
# Did it read as columns instead of rows?
|
||||
if lines_for_detection.class == Array
|
||||
lines_for_detection.first
|
||||
end
|
||||
# Carriage return without newline
|
||||
lines_for_detection = lines_for_detection.split("\x0D")
|
||||
end
|
||||
|
||||
occurrences = Hash[
|
||||
COMMON_DELIMITERS.map { |delimiter|
|
||||
[delimiter, lines_for_detection.map { |line|
|
||||
line.count(delimiter) }]
|
||||
}
|
||||
]
|
||||
|
||||
stream.rewind
|
||||
|
||||
variances = Hash.new
|
||||
@delimiter = DEFAULT_DELIMITER
|
||||
|
||||
use_variance = true
|
||||
occurrences.each { |key, values|
|
||||
if values.length > 1
|
||||
variances[key] = sample_variance(values) unless values.first == 0
|
||||
elsif values.length == 1
|
||||
# If only detected a single line of data, cannot use variance
|
||||
variances[key] = values.first * DELIMITER_WEIGHTS[key]
|
||||
use_variance = false
|
||||
else
|
||||
use_variance = false
|
||||
end
|
||||
}
|
||||
|
||||
if variances.length > 0
|
||||
if use_variance
|
||||
@delimiter = variances.sort {|a, b| a.last <=> b.last }.first.first
|
||||
else
|
||||
# Use whatever delimiter appears more and hope for the best
|
||||
@delimiter = variances.sort {|a, b| b.last <=> a.last }.first.first
|
||||
end
|
||||
end
|
||||
|
||||
@delimiter
|
||||
end
|
||||
|
||||
def self.supported?(extension)
|
||||
%w(.csv .tsv .txt).include?(extension)
|
||||
end
|
||||
|
||||
def normalize(temporary_filepath)
|
||||
|
||||
temporary_csv = CSV.open(temporary_filepath, 'w', col_sep: OUTPUT_DELIMITER, encoding: 'UTF-8')
|
||||
|
||||
CSV.open(filepath, "rb:#{encoding}", col_sep: @delimiter) do |input|
|
||||
loop do
|
||||
begin
|
||||
row = input.shift
|
||||
break unless row
|
||||
rescue CSV::MalformedCSVError
|
||||
next
|
||||
end
|
||||
temporary_csv << multiple_column(row)
|
||||
end
|
||||
end
|
||||
|
||||
# TODO: it would be nice to detect and warn the user about ignored
|
||||
# malformed rows (but probably not about malformed empty lines, such
|
||||
# as trailing \n\r\n seen in some cases)
|
||||
|
||||
temporary_csv.close
|
||||
|
||||
@delimiter = OUTPUT_DELIMITER
|
||||
rescue ArgumentError, Encoding::UndefinedConversionError, Encoding::InvalidByteSequenceError => e
|
||||
raise EncodingDetectionError
|
||||
end
|
||||
|
||||
def temporary_filepath(filename_prefix = '')
|
||||
File.join(temporary_directory, filename_prefix + File.basename(filepath))
|
||||
end
|
||||
|
||||
def csv_options
|
||||
{
|
||||
col_sep: delimiter,
|
||||
quote_char: DEFAULT_QUOTE
|
||||
}
|
||||
end
|
||||
|
||||
def needs_normalization?
|
||||
(!ACCEPTABLE_ENCODINGS.include?(encoding)) ||
|
||||
(delimiter != DEFAULT_DELIMITER) ||
|
||||
single_column?
|
||||
end
|
||||
|
||||
def single_column?
|
||||
columns = ::CSV.parse(first_line, csv_options)
|
||||
raise EmptyFileError.new if !columns.any?
|
||||
columns.first.length < 2
|
||||
end
|
||||
|
||||
def multiple_column(row)
|
||||
return row if row.length > 1
|
||||
row << nil
|
||||
end
|
||||
|
||||
def delimiter
|
||||
@delimiter
|
||||
end
|
||||
|
||||
def encoding
|
||||
return @encoding unless @encoding.nil?
|
||||
|
||||
source_file = SourceFile.new(filepath)
|
||||
if source_file.encoding
|
||||
@encoding = source_file.encoding
|
||||
else
|
||||
data = File.open(filepath, 'r')
|
||||
sample = data.read(SAMPLE_READ_LIMIT)
|
||||
data.close
|
||||
|
||||
result = CharlockHolmes::EncodingDetector.detect(sample)
|
||||
# Looks like an ICU problem https://github.com/brianmario/charlock_holmes/issues/38
|
||||
@encoding = if result.fetch(:encoding, 'UTF-8') == 'IBM424_rtl'
|
||||
DEFAULT_ENCODING
|
||||
elsif result.fetch(:confidence, 0) < ENCODING_CONFIDENCE
|
||||
DEFAULT_ENCODING
|
||||
else
|
||||
result.fetch(:encoding, DEFAULT_ENCODING)
|
||||
end
|
||||
end
|
||||
|
||||
@encoding
|
||||
rescue
|
||||
DEFAULT_ENCODING
|
||||
end
|
||||
|
||||
def first_line
|
||||
return @first_line if @first_line
|
||||
stream.rewind
|
||||
@first_line ||= stream.gets || ''
|
||||
stream.rewind
|
||||
@first_line
|
||||
end
|
||||
|
||||
def release
|
||||
@stream.close
|
||||
@stream = nil
|
||||
@first_line = nil
|
||||
self
|
||||
end
|
||||
|
||||
def stream
|
||||
@stream ||= File.open(filepath, 'rb')
|
||||
end
|
||||
|
||||
attr_reader :filepath
|
||||
alias_method :converted_filepath, :filepath
|
||||
|
||||
private
|
||||
|
||||
def generate_temporary_directory
|
||||
self.temporary_directory = Unp.new(@importer_config).generate_temporary_directory.temporary_directory
|
||||
self
|
||||
end
|
||||
|
||||
def temporary_directory
|
||||
generate_temporary_directory unless @temporary_directory
|
||||
@temporary_directory
|
||||
end
|
||||
|
||||
def sum(items_list)
|
||||
items_list.inject(0){|accum, i| accum + i }
|
||||
end
|
||||
|
||||
def mean(items_list)
|
||||
sum(items_list) / items_list.length.to_f
|
||||
end
|
||||
|
||||
def sample_variance(items_list)
|
||||
m = mean(items_list)
|
||||
sum = items_list.inject(0){|accum, i| accum + (i-m)**2 }
|
||||
sum / (items_list.length - 1).to_f
|
||||
end
|
||||
|
||||
def remove_quoted_strings(input)
|
||||
# Note that CSV quoted strings can use double quotes, `""`
|
||||
# as a way of escaping a single quote `"`
|
||||
# Since we're just removing all quoted strings, this simple
|
||||
# approach works in that case too.
|
||||
input.gsub(/"[^\\"]*"/, '')
|
||||
end
|
||||
|
||||
attr_writer :temporary_directory
|
||||
|
||||
end
|
||||
end
|
||||
end
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user