Initial commit

This commit is contained in:
zhongjin
2020-06-15 10:58:47 +08:00
commit 4f1dfe7564
8590 changed files with 1516878 additions and 0 deletions
@@ -0,0 +1,36 @@
require_relative './memory'
require_relative './redis'
module DataRepository
module Backend
class Detector
def initialize(backend_or_connection=nil)
@backend_or_connection = backend_or_connection
end #initialize
def detect
# TYPE CHECKING OMG!! SEND THE CRAFTSMANSHIPTROOPERS!!
return backend_or_connection if is_backend?
return Backend::Redis.new(backend_or_connection) if is_redis?
return default_backend
end #detect
private
attr_reader :backend_or_connection
def default_backend
Backend::Memory.new
end #default_backend
def is_backend?
!!(backend_or_connection.class.name =~ /^DataRepository::Backend/)
end #is_backend?
def is_redis?
backend_or_connection.respond_to? :zremrangebyscore
end #is_redis?
end # Detector
end # Backend
end # DataRepository
@@ -0,0 +1,27 @@
require 'set'
require 'json'
module DataRepository
module Backend
class Memory < Hash
def store(key, data, options={})
data = data.to_a if data.is_a?(Set) # OMG FIXME
super(key, JSON.parse(data.to_json))
end
def fetch(key, options={})
super key
end
def exists?(key)
self.has_key?(key)
end
# Not supported, so just call data
def transaction(&block)
block.call
end
end
end
end
+63
View File
@@ -0,0 +1,63 @@
require 'redis'
require_relative 'redis/set'
require_relative 'redis/string'
module DataRepository
module Backend
class Redis
HANDLERS = {
'set' => Backend::Redis::Set,
'string' => Backend::Redis::String
}
def initialize(redis=::Redis.new)
@redis = redis
end #initialize
def store(key, data, options={})
persister_for(data).new(redis).store(key, data)
expire_in(options.fetch(:expiration, nil), key)
end #store
def fetch(key)
return nil unless redis.exists(key)
retriever_for(key).new(redis).fetch(key)
end #fetch
def keys
redis.keys
end #keys
def exists?(key)
redis.exists(key)
end #exists?
def delete(key)
redis.del(key)
end #delete
# Not supported, so just call data
def transaction(&block)
block.call
end
private
attr_reader :redis
def expire_in(seconds, key)
!!seconds && redis.expire(key, seconds)
end #expire_in
def retriever_for(key)
HANDLERS.fetch(redis.type(key), Backend::Redis::String)
end #retriever_for
def persister_for(data)
HANDLERS.fetch(data.class.to_s.downcase, Backend::Redis::String)
end #persister_for(data)
end # Redis
end # Backend
end # DataRepository
@@ -0,0 +1,32 @@
require 'redis'
module DataRepository
module Backend
class Redis
class Set
def initialize(redis=Redis.new)
@redis = redis
end #initialize
def store(key, data)
workaround_until_resque_supports_latest_redis_gem(key, data)
end #store
def fetch(key)
redis.smembers key
end #fetch
private
attr_reader :redis
def workaround_until_resque_supports_latest_redis_gem(key, data)
redis.multi do
data.to_a.each { |item| redis.sadd(key, item) }
end
end #workaround_until_resque_supports_latest_redis_gem
end # Set
end # Redis
end # Backend
end # DataRepository
@@ -0,0 +1,27 @@
require 'json'
require 'redis'
module DataRepository
module Backend
class Redis
class String
def initialize(redis=Redis.new)
@redis = redis
end #initialize
def store(key, data)
redis.set key, data.to_json
end #store
def fetch(key)
JSON.parse redis.get(key)
end #fetch
private
attr_reader :redis
end # String
end # Redis
end # Backend
end # DataRepository
+135
View File
@@ -0,0 +1,135 @@
require 'sequel'
require 'uuidtools'
module DataRepository
module Backend
class Sequel
PAGE = 1
PER_PAGE = 300
ARRAY_RE = %r{\[.*\]}
def initialize(db, relation=nil)
@db = db
@relation = relation.to_sym
::Sequel.extension(:pagination)
::Sequel.extension(:connection_validator)
@db.extension :pg_array if postgres?(@db)
end
def collection(filters={}, available_filters=[])
apply_filters(db[relation], filters, available_filters)
end
def store(key, data={})
naive_upsert_exposed_to_race_conditions(data)
end
def fetch(value, key=nil)
if key.nil?
parse( db[relation].where(id: value).first )
else
parse( db[relation].where(key.to_sym => value).first )
end
end
def delete(key)
db[relation].where(id: key).delete
end
def next_id
UUIDTools::UUID.timestamp_create
end
def apply_filters(dataset, filters={}, available_filters=[])
return dataset if filters.nil? || filters.empty?
available_filters = symbolize_elements(available_filters)
filters = symbolize_keys(filters).select { |key, value|
available_filters.include?(key)
} unless available_filters.empty?
dataset.where(filters)
end
def paginate(dataset, filter={}, record_count=nil)
page, per_page = pagination_params_from(filter)
dataset.paginate(page, per_page, record_count)
end
def transaction(&block)
db.transaction(&block)
end
private
attr_reader :relation, :db
def naive_upsert_exposed_to_race_conditions(data={})
data = send("serialize_for_#{backend_type}", data)
insert(data) unless update(data)
end
def insert(data={})
db[relation].insert(data)
end
def update(data={})
db[relation].where(id: data.fetch(:id)).update(data) != 0
end
def serialize_for_postgres(data)
Hash[
data.map { |key, value|
value = ::Sequel.pg_array(value) if value.is_a?(Array) && !value.empty?
[key, value]
}
]
end
def serialize_for_other_database(data={})
Hash[
data.map { |key, value|
value = value.to_s if value.is_a?(Array)
[key, value]
}
]
end
def backend_type
postgres?(db) ? :postgres : :other_database
end
def postgres?(db)
db.database_type == :postgres
end
def parse(attributes={})
return unless attributes
return attributes if postgres?(db)
Hash[
attributes.map do |key, value|
value = JSON.parse(value) if value =~ ARRAY_RE
[key, value]
end
]
end
def symbolize_elements(array=[])
array.map { |k| k.to_sym}
end
def symbolize_keys(hash={})
Hash[ hash.map { |k, v| [k.to_sym, v] } ]
end
def pagination_params_from(filter)
page = (filter.delete(:page) || PAGE).to_i
per_page = (filter.delete(:per_page) || PER_PAGE).to_i
[page, per_page]
end
end
end
end
@@ -0,0 +1,78 @@
require 'fileutils'
module DataRepository
module Filesystem
class Local
DEFAULT_PREFIX = File.join(File.dirname(__FILE__), '..', 'tmp')
def initialize(base_directory=DEFAULT_PREFIX)
@base_directory = base_directory
end
def create_base_directory
FileUtils.mkpath @base_directory unless exists? @base_directory
end
def store(path, data)
FileUtils.mkpath(File.dirname(fullpath_for(path)))
File.open(fullpath_for(path), 'wb') do |file|
data.rewind if data.eof?
if data.respond_to?(:bucket)
data.read { |chunk| file.write(chunk) }
else
chunk = data.gets
while chunk
file.write(chunk)
chunk = data.gets
end
end
end
path
end
def fetch(path)
File.open(fullpath_for(path), 'r')
end
def exists?(path)
File.exists?(fullpath_for(path))
end
# Use from controlled environments always
def remove(path)
if exists?(path)
File.delete(fullpath_for(path))
end
end
def fullpath_for(path)
File.join(base_directory, path)
end
private
attr_reader :base_directory
def targets_for(path)
fullpath = fullpath_for(path)
[
Dir.glob(fullpath),
Dir.glob("#{fullpath}/*"),
Dir.glob("#{fullpath}/**/*")
].flatten
.uniq
.delete_if { |entry| dot_directory?(entry) }
end
def dot_directory?(path)
path == '.' || path == '..'
end
def relative_path_for(path, base_directory)
(path.split('/') - base_directory.split('/')).join('/')
end
end
end
end
+49
View File
@@ -0,0 +1,49 @@
require 'uuidtools'
require_relative 'backend/detector'
require_relative 'backend/memory'
module DataRepository
def self.new(backend_or_connection=nil)
backend = Backend::Detector.new(backend_or_connection).detect
Repository.new(backend)
end # DataRepository.new
class Repository
def initialize(storage=Backend::Memory.new)
@storage = storage
end
def backend
@storage
end
def store(key, data, options={})
storage.store(key.to_s, data, options)
end
def fetch(key)
storage.fetch(key.to_s)
end
def delete(key)
storage.delete(key.to_s)
end
def exists?(key)
storage.exists?(key)
end
def keys
storage.keys
end
def next_id
UUIDTools::UUID.timestamp_create
end
private
attr_reader :storage
end # Handler
end # DataRepository
@@ -0,0 +1,21 @@
require 'minitest/autorun'
require 'redis'
require_relative '../spec_helper'
require_relative '../../repository'
describe DataRepository do
describe 'DataRepository.new' do
it 'instantiates a repository with a memory backend' do
repository = DataRepository.new
repository.must_be_instance_of DataRepository::Repository
repository.backend.must_be_instance_of DataRepository::Backend::Memory
end
it 'detects the backend if a repository or DB connection is passed' do
repository = DataRepository.new(Redis.new)
repository.must_be_instance_of DataRepository::Repository
repository.backend.must_be_instance_of DataRepository::Backend::Redis
end
end # DataRepository.new
end # DataRepository
@@ -0,0 +1,30 @@
require 'minitest/autorun'
require 'redis'
require_relative '../../spec_helper'
require_relative '../../../backend/detector'
include DataRepository
describe Backend::Detector do
describe '#detect' do
it 'returns a memory backend if no connection or backend passed' do
Backend::Detector.new.detect.must_be_instance_of Backend::Memory
end
it "returns the passed object if it's a backend" do
memory_backend = Backend::Memory.new
detector = Backend::Detector.new(memory_backend)
detector.detect.must_be_instance_of Backend::Memory
redis_backend = Backend::Redis.new
detector = Backend::Detector.new(redis_backend)
detector.detect.must_be_instance_of Backend::Redis
end
it 'returns a redis backend if a redis connection is passed' do
detector = Backend::Detector.new(Redis.new)
detector.detect.must_be_instance_of Backend::Redis
end
end #detect
end # Backend::Detector
@@ -0,0 +1,81 @@
require 'minitest/autorun'
require 'ostruct'
require_relative '../../spec_helper'
require_relative '../../../repository'
require_relative '../../../backend/memory'
include DataRepository
describe Repository do
before do
@repository = Repository.new(Backend::Memory.new)
end
describe '#store' do
it 'persists a data structure in the passed key' do
data = { id: 5 }
key = data.fetch(:id)
@repository.keys.wont_include key.to_s
@repository.store(key, data)
@repository.keys.must_include key.to_s
end
it 'stringifies symbols in the persisted data structure' do
data = { id: 5 }
key = data.fetch(:id)
@repository.store(key, data)
retrieved_data = @repository.fetch(key)
retrieved_data.keys.wont_include :id
retrieved_data.keys.must_include 'id'
end
end #store
describe '#fetch' do
it 'retrieves a data structure from a key' do
data = { id: 5 }
key = data.fetch(:id)
@repository.store(key, data)
retrieved_data = @repository.fetch(key.to_s)
retrieved_data.fetch('id').must_equal data.fetch(:id)
end
end #fetch
describe '#delete' do
it 'deletes a key' do
data = { id: 5 }
key = data.fetch(:id)
@repository.store(key, data)
@repository.fetch(key.to_s).wont_be_nil
@repository.delete(key)
lambda { @repository.fetch(key.to_s) }.must_raise KeyError
end
end #delete
describe '#keys' do
it 'returns all stored keys, stringified' do
data = { id: 5 }
key = data.fetch(:id)
@repository.store(key, data)
@repository.keys.must_equal [key.to_s]
end
end #keys
describe '#exists?' do
it 'returns if key exists' do
data = { id: 5 }
key = data.fetch(:id)
@repository.exists?(key.to_s).must_equal false
@repository.store(key, data)
@repository.exists?(key.to_s).must_equal true
end
end #exists?
end # Repository
@@ -0,0 +1,111 @@
require 'minitest/autorun'
require 'set'
require_relative '../../spec_helper'
require_relative '../../../backend/redis'
require_relative '../../../repository'
include DataRepository
describe Backend::Redis do
before do
@connection = Redis.new
@connection.select 8
@connection.flushdb
storage = Backend::Redis.new(@connection)
@repository = Repository.new(storage)
end
describe '#store' do
it 'persists a data structure in the passed key' do
data = { id: 5 }
key = data.fetch(:id)
@repository.keys.wont_include key.to_s
@repository.store(key, data)
@repository.keys.must_include key.to_s
set = Set.new
set.add 1
set.add 2
@repository.store('bogus_key', set)
@repository.fetch('bogus_key').must_be_kind_of Enumerable
end
it 'stringifies symbols in the persisted data structe' do
data = { id: 5 }
key = data.fetch(:id)
@repository.store(key, data)
retrieved_data = @repository.fetch(key)
retrieved_data.keys.wont_include :id
retrieved_data.keys.must_include 'id'
end
it 'sets key expiration in seconds if expiration option passed' do
data = { id: 5 }
key = data.fetch(:id)
expiration = 1
@repository.store(key, data, expiration: expiration)
retrieved_data = @repository.fetch(key)
retrieved_data.keys.must_include 'id'
@connection.get(key).wont_be_nil
sleep(expiration.to_f + 1.0 / 1.0)
@connection.get(key).must_be_nil
end
end #store
describe '#fetch' do
it 'retrieves a data structure from a key' do
data = { id: 5 }
key = data.fetch(:id)
@repository.store(key, data)
retrieved_data = @repository.fetch(key)
retrieved_data.fetch('id').must_equal data.fetch(:id)
end
it 'returns nil if key does not exist' do
@repository.fetch('non_existent_key').must_equal nil
end
end #fetch
describe '#delete' do
it 'deletes a key' do
data = { id: 5 }
key = data.fetch(:id)
@repository.store(key, data)
@repository.fetch(key).wont_be_nil
@repository.delete(key)
@repository.fetch(key).must_be_nil
end
end #delete
describe '#keys' do
it 'returns all stored keys, stringified' do
data = { id: 5 }
key = data.fetch(:id)
@repository.store(key, data)
@repository.keys.must_equal [key.to_s]
end
end #keys
describe '#exists?' do
it 'returns if key exists' do
data = { id: 5 }
key = data.fetch(:id)
@repository.exists?(key.to_s).must_equal false
@repository.store(key, data)
@repository.exists?(key.to_s).must_equal true
end
end #exists?
end # Backend::Redis
@@ -0,0 +1,104 @@
require_relative '../../../backend/sequel'
require_relative '../../../../../app/models/visualization/member'
include CartoDB
describe DataRepository::Backend::Sequel do
before do
db = SequelRails.connection
db.create_table :visualizations do
UUID :id, primary_key: true
String :name
String :display_name
String :title
String :description
String :license
String :source
String :tags
String :map_id
String :active_layer_id
String :type
String :privacy
String :encrypted_password
String :password_salt
UUID :permission_id
Boolean :locked
String :parent_id
String :kind
String :prev_id
String :next_id
String :slide_transition_options
String :active_child
end
db.create_table :overlays do
String :id, null: false, primary_key: true
Integer :order, null: false
String :options, text: true
String :type
String :visualization_id, index: true
end
Visualization.repository = DataRepository::Backend::Sequel.new(db, :visualizations)
Overlay.repository = DataRepository::Backend::Sequel.new(db, :overlays)
end
describe '#store' do
it 'inserts a visualization' do
member = Visualization::Member.new(
name: 'visualization 1',
tags: ['foo', 'bar']
)
member.store
rehydrated_member = Visualization::Member.new(id: member.id)
rehydrated_member.fetch
rehydrated_member.name.should eq member.name
end
it 'updates the visualization if existing' do
member = Visualization::Member.new(
name: 'visualization 1',
tags: ['foo', 'bar']
)
member.store
Visualization.repository.collection(id: member.id).to_a.size.should eq 1
member.store
Visualization.repository.collection(id: member.id).to_a.size.should eq 1
Visualization::Member.new(id: member.id).fetch.store
Visualization.repository.collection(id: member.id).to_a.size.should eq 1
end
end
describe '#delete' do
it 'deletes a visualization from persistence' do
member = Visualization::Member.new(
name: 'visualization 1',
tags: ['foo', 'bar']
).store
id = member.id
Visualization.repository.fetch(id).nil?.should eq false
member.delete
Visualization.repository.fetch(id).nil?.should eq true
end
end
describe '#collection' do
it 'gets a collection of records using the passed filter' do
Visualization::Member.new(
name: 'visualization 1',
map_id: 1
).store
Visualization::Member.new(
name: 'visualization 2',
map_id: 1
).store
records = Visualization.repository.collection(map_id: 1)
records.to_a.size.should eq 2
end
end
end
@@ -0,0 +1,53 @@
require 'minitest/autorun'
require 'stringio'
require_relative '../../../filesystem/local'
include DataRepository::Filesystem
describe Local do
before do
@data = StringIO.new(Time.now.to_f.to_s)
@path = File.join( (0..2).map { Time.now.to_f.to_s } )
@prefix = File.join(Local::DEFAULT_PREFIX, Time.now.to_i.to_s)
end
after do
@data.close
FileUtils.rmtree(@prefix)
FileUtils.rmtree(Local::DEFAULT_PREFIX)
end
describe '#initialize' do
it 'sets the storage prefix to Local::DEFAULT_PREFIX by default' do
File.exists?( File.join(Local::DEFAULT_PREFIX, @path) ).must_equal false
path = Local.new.store(@path, @data)
File.exists?( File.join(Local::DEFAULT_PREFIX, @path) ).must_equal true
end
end #initialize
describe '#store' do
it 'stores data in the specified path' do
filesystem = Local.new(@prefix)
path = filesystem.store(@path, @data)
@data.rewind
stored_data = File.open( File.join(@prefix, path) )
stored_data.read.must_equal @data.read
end
end #store
describe '#fetch' do
it 'retrieves data from the specified path' do
filesystem = Local.new(@prefix)
path = filesystem.store(@path, @data)
@data.rewind
stored_data = Local.new(@prefix).fetch(path)
stored_data.read.must_equal @data.read
end
end #fetch
end # Local
@@ -0,0 +1,123 @@
require 'minitest/autorun'
require_relative '../../../structures/collection'
include DataRepository
describe Collection do
before do
@repository = DataRepository::Repository.new
@dummy_class = Class.new do
attr_accessor :id
def initialize(arguments={}); self.id = arguments.fetch(:id); end
def fetch; self; end
def to_hash; { id: id }; end
def ==(other); id.to_s == other.id.to_s; end
end
@defaults = { repository: @repository, member_class: @dummy_class}
end
describe '#add' do
it 'adds a member to the collection' do
member = @dummy_class.new(id: 1)
collection = Collection.new(@defaults)
collection.add(member)
collection.to_a.must_include member
end
end
describe '#delete' do
it 'deletes a member from the collection' do
member = @dummy_class.new(id: 1)
collection = Collection.new(@defaults)
collection.add(member)
collection.delete(member)
collection.to_a.wont_include member
end
end #delete
describe '#each' do
it 'yields members of the collection as the initialized member_class' do
member = @dummy_class.new(id: 1)
collection = Collection.new(@defaults)
collection.add(member)
collection.store
rehydrated_collection =
Collection.new(@defaults.merge(signature: collection.signature))
rehydrated_collection.fetch
rehydrated_collection.to_a.first.must_be_instance_of @dummy_class
end
it 'returns an enumerator if no block given' do
member = @dummy_class.new(id: 1)
collection = Collection.new({ repository: @repository })
collection.add(member)
collection.store
rehydrated_collection =
Collection.new(@defaults.merge(signature: collection.signature))
rehydrated_collection.fetch
enumerator = rehydrated_collection.each
enumerator.next.must_be_instance_of @dummy_class
end
end #each
describe '#fetch' do
it 'resets the collection with data from the data repository' do
member1 = @dummy_class.new(id: 1)
member2 = @dummy_class.new(id: 2)
collection = Collection.new(@defaults)
collection.add(member1)
collection.store
rehydrated_collection =
Collection.new(@defaults.merge(signature: collection.signature))
rehydrated_collection.add(member2)
rehydrated_collection.to_a.must_include(member2)
rehydrated_collection.to_a.wont_include(member1)
rehydrated_collection.fetch
rehydrated_collection.to_a.must_include(member1)
rehydrated_collection.to_a.wont_include(member2)
end
it 'empties the collection if it was not persisted to the repository' do
member = @dummy_class.new(id: 1)
collection = Collection.new(@defaults)
collection.add(member)
collection.to_a.length.must_equal 1
collection.fetch
collection.to_a.must_be_empty
end
end #fetch
describe '#store' do
it 'persists the collection to the data repository' do
member = @dummy_class.new(id: 1)
collection = Collection.new(@defaults)
collection.add(member)
collection.store
rehydrated_collection =
Collection.new(@defaults.merge(signature: collection.signature))
rehydrated_collection.fetch
rehydrated_collection.map { |member| member.id }.must_include member.id
end
end #store
describe '#to_json' do
it 'renders a JSON representation of the collection' do
member = @dummy_class.new(id: 1)
collection = Collection.new(@defaults)
collection.add(member)
collection.store
representation = JSON.parse(collection.to_json)
representation.size.must_equal 1
representation.first.fetch('id').must_equal member.id
end
end #to_json
end # Collection
@@ -0,0 +1,75 @@
require 'ostruct'
require 'set'
require_relative '../repository'
module DataRepository
class Collection
include Enumerable
INTERFACE = %w{ signature add delete store fetch each to_json repository count } + Enumerable.instance_methods
attr_reader :signature
attr_accessor :storage
def initialize(arguments={})
@storage = Set.new
@member_class = arguments.fetch(:member_class, OpenStruct)
@repository = arguments.fetch(:repository, Repository.new)
@signature = arguments.fetch(:signature, @repository.next_id)
end #initialize
def add(member)
storage.add(member)
self
end #add
def delete(member)
storage.delete(member)
self
end #delete
def delete_if(&block)
storage.delete_if(&block)
self
end
def each(&block)
return storage.each(&block)
Enumerator.new(self, :each)
end #each
def fetch
self.storage = Set[*repository.fetch(signature)].map do |attributes|
puts attributes.inspect
member_class.new(attributes)
end
self
rescue => exception
storage.clear
self
end #fetch
def store
repository.store(signature, storage.map(&:id).to_a)
self
end #store
def to_json(*args)
map { |member| member.to_hash }.to_json(*args)
end #to_json
def count
storage.count
end #count
private
attr_reader :repository, :member_class
def members
storage.each { |member_id| yield member_class.new(id: member_id).fetch }
end #members
end # Collection
end # DataRepository
@@ -0,0 +1,46 @@
require 'active_support/time'
require_relative 'service_usage_metrics'
module CartoDB
# The purpose of this class is to encapsulate storage of usage metrics.
# This shall be used for billing, quota checking and metrics.
class GeocoderUsageMetrics < ServiceUsageMetrics
VALID_METRICS = [
:total_requests,
:failed_responses,
:success_responses,
:empty_responses,
].freeze
VALID_SERVICES = [
:geocoder_internal,
:geocoder_here,
:geocoder_google,
:geocoder_cache,
:geocoder_mapzen,
:geocoder_mapbox,
:geocoder_tomtom
].freeze
GEOCODER_KEYS = {
"heremaps" => :geocoder_here,
"google" => :geocoder_google,
"mapzen" => :geocoder_mapzen,
"mapbox" => :geocoder_mapbox,
"tomtom" => :geocoder_tomtom
}.freeze
def initialize(username, orgname = nil, redis=$geocoder_metrics)
super(username, orgname, redis)
end
protected
def check_valid_data(service, metric)
raise ArgumentError.new('Invalid service') unless VALID_SERVICES.include?(service)
raise ArgumentError.new('Invalid metric') unless VALID_METRICS.include?(metric)
end
end
end
@@ -0,0 +1,42 @@
require 'active_support/time'
require_relative 'service_usage_metrics'
module CartoDB
# The purpose of this class is to encapsulate storage of usage metrics.
# This shall be used for billing, quota checking and metrics.
class IsolinesUsageMetrics < ServiceUsageMetrics
VALID_METRICS = [
:total_requests,
:failed_responses,
:success_responses,
:empty_responses,
:isolines_generated
].freeze
VALID_SERVICES = [
:here_isolines,
:mapzen_isolines,
:mapbox_isolines,
:tomtom_isolines
].freeze
ISOLINES_KEYS = {
"heremaps" => :here_isolines,
"mapzen" => :mapzen_isolines,
"mapbox" => :mapbox_isolines,
"tomtom" => :tomtom_isolines
}.freeze
def initialize(username, orgname = nil, redis=$geocoder_metrics)
super(username, orgname, redis)
end
protected
def check_valid_data(service, metric)
raise ArgumentError.new('Invalid service') unless VALID_SERVICES.include?(service)
raise ArgumentError.new('Invalid metric') unless VALID_METRICS.include?(metric)
end
end
end
@@ -0,0 +1,31 @@
require 'active_support/time'
require_relative 'service_usage_metrics'
module CartoDB
# The purpose of this class is to encapsulate storage of usage metrics.
# This shall be used for billing, quota checking and metrics.
class ObservatoryGeneralUsageMetrics < ServiceUsageMetrics
VALID_METRICS = [
:total_requests,
:failed_responses,
:success_responses,
:empty_responses
].freeze
VALID_SERVICES = [
:obs_general
].freeze
def initialize(username, orgname = nil, redis = $geocoder_metrics)
super(username, orgname, redis)
end
protected
def check_valid_data(service, metric)
raise ArgumentError.new('Invalid service') unless VALID_SERVICES.include?(service)
raise ArgumentError.new('Invalid metric') unless VALID_METRICS.include?(metric)
end
end
end
@@ -0,0 +1,31 @@
require 'active_support/time'
require_relative 'service_usage_metrics'
module CartoDB
# The purpose of this class is to encapsulate storage of usage metrics.
# This shall be used for billing, quota checking and metrics.
class ObservatorySnapshotUsageMetrics < ServiceUsageMetrics
VALID_METRICS = [
:total_requests,
:failed_responses,
:success_responses,
:empty_responses
].freeze
VALID_SERVICES = [
:obs_snapshot
].freeze
def initialize(username, orgname = nil, redis = $geocoder_metrics)
super(username, orgname, redis)
end
protected
def check_valid_data(service, metric)
raise ArgumentError.new('Invalid service') unless VALID_SERVICES.include?(service)
raise ArgumentError.new('Invalid metric') unless VALID_METRICS.include?(metric)
end
end
end
@@ -0,0 +1,39 @@
require 'active_support/time'
require_relative 'service_usage_metrics'
module CartoDB
# The purpose of this class is to encapsulate storage of usage metrics.
# This shall be used for billing, quota checking and metrics.
class RoutingUsageMetrics < ServiceUsageMetrics
VALID_METRICS = [
:total_requests,
:failed_responses,
:success_responses,
:empty_responses
].freeze
VALID_SERVICES = [
:routing_mapzen,
:routing_mapbox,
:routing_tomtom
].freeze
ROUTING_KEYS = {
"mapzen" => :routing_mapzen,
"mapbox" => :routing_mapbox,
"tomtom" => :routing_tomtom
}.freeze
def initialize(username, orgname = nil, redis = $geocoder_metrics)
super(username, orgname, redis)
end
protected
def check_valid_data(service, metric)
raise ArgumentError.new('Invalid service') unless VALID_SERVICES.include?(service)
raise ArgumentError.new('Invalid metric') unless VALID_METRICS.include?(metric)
end
end
end
@@ -0,0 +1,91 @@
require 'active_support/time'
require_relative '../../../lib/carto/metrics/usage_metrics_interface'
module CartoDB
# The purpose of this class is to encapsulate storage of usage metrics.
# This shall be used for billing, quota checking and metrics.
class ServiceUsageMetrics < Carto::Metrics::UsageMetricsInterface
def initialize(username, orgname = nil, redis=$geocoder_metrics)
@username = username
@orgname = orgname
@redis = redis
end
def incr(service, metric, amount = 1, date = DateTime.current)
check_valid_data(service, metric)
assert_valid_amount(amount)
return if amount == 0
# TODO We could add EXPIRE command to add TTL to the keys
if !@orgname.nil?
@redis.zincrby("#{org_key_prefix(service, metric, date)}", amount, "#{date_day(date)}")
end
@redis.zincrby("#{user_key_prefix(service, metric, date)}", amount, "#{date_day(date)}")
end
def get(service, metric, date = DateTime.current)
check_valid_data(service, metric)
total = 0
if !@orgname.nil?
total += @redis.zscore(org_key_prefix(service, metric, date), date_day(date)) || 0
else
total += @redis.zscore(user_key_prefix(service, metric, date), date_day(date)) || 0
end
total
end
def get_sum_by_date_range(service, metric, date_from, date_to)
get_date_range(service, metric, date_from, date_to).values.reduce(:+)
end
def get_date_range(service, metric, date_from, date_to)
check_valid_data(service, metric)
ret = {}
month_values = {}
(date_from..date_to).each do |date|
year_month_key = date_year_month(date)
if month_values[year_month_key].nil?
key_prefix = @orgname.nil? ? user_key_prefix(service, metric, date) : org_key_prefix(service, metric, date)
month_values[year_month_key] = @redis.zrange(key_prefix, 0, -1, with_scores: true).to_h
end
ret[date] = month_values[year_month_key][date_day(date)] || 0
end
ret
end
protected
def check_valid_data(_service, _metric)
raise NotImplementedError.new("You must implement check_valid_data in your metrics class.")
end
private
def user_key_prefix(service, metric, date)
"user:#{@username}:#{service}:#{metric}:#{date_year_month(date)}"
end
def org_key_prefix(service, metric, date)
"org:#{@orgname}:#{service}:#{metric}:#{date_year_month(date)}"
end
def date_day(date)
date.strftime('%d')
end
def date_year_month(date)
date.strftime('%Y%m')
end
def assert_valid_amount(amount)
raise ArgumentError.new('Invalid metric amount') if amount.nil? || amount < 0
end
end
end
@@ -0,0 +1,184 @@
require_relative '../../lib/service_usage_metrics'
require 'mock_redis'
require_relative '../../../../spec/rspec_configuration'
describe CartoDB::ServiceUsageMetrics do
class DummyServiceUsageMetrics < CartoDB::ServiceUsageMetrics
VALID_METRICS = [:dummy_metric].freeze
VALID_SERVICES = [:dummy_service].freeze
def check_valid_data(service, metric)
raise ArgumentError.new('Invalid service') unless VALID_SERVICES.include?(service)
raise ArgumentError.new('Invalid metric') unless VALID_METRICS.include?(metric)
end
end
before(:each) do
@redis_mock = MockRedis.new
@usage_metrics = DummyServiceUsageMetrics.new('rtorre', 'team', @redis_mock)
end
describe 'Read quota info from redis with zero padding' do
it 'reads standard zero padded keys' do
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201606', 1543, '01')
@usage_metrics.get(:dummy_service, :dummy_metric, Date.new(2016, 6, 1)).should eq 1543
end
it "does not request redis twice when there's no need" do
@redis_mock.expects(:zscore).once.with('org:team:dummy_service:dummy_metric:201606', '20').returns(3141592)
@usage_metrics.get(:dummy_service, :dummy_metric, Date.new(2016, 6, 20)).should eq 3141592
end
it "returns zero when there's no consumption" do
@usage_metrics.get(:dummy_service, :dummy_metric, Date.new(2016, 6, 20)).should eq 0
end
end
describe :assert_valid_amount do
it 'passes when fed with a positive integer' do
@usage_metrics.send(:assert_valid_amount, 42).should eq nil
end
it 'validates that the amount passed cannot be nil' do
expect {
@usage_metrics.send(:assert_valid_amount, nil)
}.to raise_exception(ArgumentError, 'Invalid metric amount')
end
it 'validates that the amount passed cannot be negative' do
expect {
@usage_metrics.send(:assert_valid_amount, -42)
}.to raise_exception(ArgumentError, 'Invalid metric amount')
end
it 'validates that the amount passed can actually be zero' do
@usage_metrics.send(:assert_valid_amount, 0).should eq nil
end
end
describe :incr do
it 'validates that the amount passed can actually be zero' do
@usage_metrics.incr(:dummy_service, :dummy_metric, _amount = 0)
@usage_metrics.get(:dummy_service, :dummy_metric).should eq 0
end
end
describe '#get_sum_by_date_range' do
it 'gets a sum of the zscores stored in a given date range' do
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 1, _day = '20')
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 2, _day = '21')
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 3, _day = '22')
@usage_metrics.get_sum_by_date_range(:dummy_service,
:dummy_metric,
Date.new(2017, 3, 20),
Date.new(2017, 3, 22)).should eq 6
end
it 'gracefully deals with days without record' do
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 1, _day = '20')
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 2, _day = '21')
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 3, _day = '22')
@usage_metrics.get_sum_by_date_range(:dummy_service,
:dummy_metric,
Date.new(2017, 3, 15),
Date.new(2017, 3, 22)).should eq 6
end
it 'gracefully deals with months not stored in redis' do
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 1, _day = '20')
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 2, _day = '21')
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 3, _day = '22')
@usage_metrics.get_sum_by_date_range(:dummy_service,
:dummy_metric,
Date.new(2017, 2, 15),
Date.new(2017, 3, 22)).should eq 6
end
it 'performs just one request/month to redis' do
@redis_mock.expects(:zrange).twice
@usage_metrics.get_sum_by_date_range(:dummy_service,
:dummy_metric,
Date.new(2017, 2, 15),
Date.new(2017, 3, 24))
end
it 'returns zero when there are no records' do
@usage_metrics.get_sum_by_date_range(:dummy_service,
:dummy_metric,
Date.new(2017, 2, 15),
Date.new(2017, 3, 22)).should eq 0
end
end
describe '#get_date_range' do
it 'gets a hash of date => value pairs' do
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 1, _day = '20')
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 2, _day = '21')
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 3, _day = '22')
expected = {
Date.new(2017, 3, 20) => 1,
Date.new(2017, 3, 21) => 2,
Date.new(2017, 3, 22) => 3
}
@usage_metrics.get_date_range(:dummy_service,
:dummy_metric,
Date.new(2017, 3, 20),
Date.new(2017, 3, 22)).should eq expected
end
it 'gracefully deals with days without record' do
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 1, _day = '20')
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 2, _day = '21')
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 3, _day = '22')
expected = {
Date.new(2017, 3, 18) => 0,
Date.new(2017, 3, 19) => 0,
Date.new(2017, 3, 20) => 1,
Date.new(2017, 3, 21) => 2,
Date.new(2017, 3, 22) => 3
}
@usage_metrics.get_date_range(:dummy_service,
:dummy_metric,
Date.new(2017, 3, 18),
Date.new(2017, 3, 22)).should eq expected
end
it 'gracefully deals with months not stored in redis' do
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 1, _day = '01')
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 2, _day = '02')
@redis_mock.zincrby('org:team:dummy_service:dummy_metric:201703', _amount = 3, _day = '03')
expected = {
Date.new(2017, 2, 27) => 0,
Date.new(2017, 2, 28) => 0,
Date.new(2017, 3, 1) => 1,
Date.new(2017, 3, 2) => 2,
Date.new(2017, 3, 3) => 3
}
@usage_metrics.get_date_range(:dummy_service,
:dummy_metric,
Date.new(2017, 2, 27),
Date.new(2017, 3, 3)).should eq expected
end
it 'performs just one request/month to redis' do
@redis_mock.expects(:zrange).twice
@usage_metrics.get_date_range(:dummy_service,
:dummy_metric,
Date.new(2017, 2, 15),
Date.new(2017, 3, 24))
end
it 'returns zero when there are no records' do
expected = {
Date.new(2017, 2, 28) => 0,
Date.new(2017, 3, 1) => 0
}
@usage_metrics.get_date_range(:dummy_service,
:dummy_metric,
Date.new(2017, 2, 28),
Date.new(2017, 3, 1)).should eq expected
end
end
end
+16
View File
@@ -0,0 +1,16 @@
require 'zlib'
require_relative './datasources/base'
require_relative './datasources/base_oauth'
require_relative './datasources/base_file_stream'
require_relative './datasources/base_direct_stream'
require_relative './datasources/exceptions'
require_relative './datasources/url/public_url'
require_relative './datasources/url/dropbox'
require_relative './datasources/url/gdrive'
require_relative './datasources/url/instagram_oauth'
require_relative './datasources/url/mailchimp'
require_relative './datasources/url/arcgis'
require_relative './datasources/search/twitter'
require_relative './datasources/datasources_factory'
require_relative './datasources/util/csv_file_dumper'
require_relative './datasources/decorators/factory'
@@ -0,0 +1,141 @@
module CartoDB
module Datasources
class Base
# .csv
FORMAT_CSV = 'csv'
# .xls .xlsx
FORMAT_EXCEL = 'xls'
# .GPX
FORMAT_GPX = 'gpx'
# .KML
FORMAT_KML = 'kml'
# .png
FORMAT_PNG = 'png'
# .jpg .jpeg
FORMAT_JPG = 'jpg'
# .svg
FORMAT_SVG = 'svg'
# .zip
FORMAT_COMPRESSED = 'zip'
# If data size cannot be determined, this will be returned as its size in the item metadata
NO_CONTENT_SIZE_PROVIDED = 0
def initialize(*args)
@logger = nil
end
# Small helper method to know if metadata includes a valid resource size value or not
# @param resource_metadata Hash { :size, ... }
# @return bool
def has_resource_size?(resource_metadata)
resource_metadata[:size] && resource_metadata[:size] > NO_CONTENT_SIZE_PROVIDED
end
# Factory method
# @param config {}
# @return mixed
def get_new(config)
raise 'To be implemented in child classes'
end
# If will provide a url to download the resource, or requires calling get_resource()
# @return bool
def providers_download_url?
raise 'To be implemented in child classes'
end
# If will provide the url http response code
# @return string
def get_http_response_code
raise 'To be implemented in child classes'
end
# Perform the listing and return results
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
# @return [ { :id, :title, :url, :service, :filename, :checksum, :size } ]
def get_resources_list(filter={})
raise 'To be implemented in child classes'
end
# Retrieves a resource and returns its contents
# @param id string
# @return mixed
def get_resource(id)
raise 'To be implemented in child classes'
end
# @param id string
# @return Hash
def get_resource_metadata(id)
raise 'To be implemented in child classes'
end
# Retrieves current filters
# @return {}
def filter
raise 'To be implemented in child classes'
end
# Sets current filters
# @param filter_data {}
def filter=(filter_data={})
raise 'To be implemented in child classes'
end
# Log a message
# @param message String
def log(message)
puts message if @logger.nil?
@logger.append(message) unless @logger.nil?
end
# @param logger Mixed|nil Set or unset the logger
def logger=(logger=nil)
@logger = logger
end
# Just return datasource name
# @return string
def to_s
raise 'To be implemented in child classes'
end
# If this datasource accepts a data import instance
# @return Boolean
def persists_state_via_data_import?
raise 'To be implemented in child classes'
end
def set_audit_to_completed(table_id = nil)
raise 'To be implemented in child classes'
end
def set_audit_to_failed
raise 'To be implemented in child classes'
end
# @return Hash
def get_audit_stats
raise 'To be implemented in child classes'
end
# Stores the data import item instance to use/manipulate it
# @param value DataImport
def data_import_item=(value)
raise 'To be implemented in child classes'
end
# If true, a single resource id might return >1 subresources (each one spawning a table)
# @param id String
# @return Bool
def multi_resource_import_supported?(id)
false
end
private_class_method :new
end
end
end
@@ -0,0 +1,25 @@
require_relative 'base'
module CartoDB
module Datasources
# Performs streaming from the datasource directly to the caller in batches
class BaseDirectStream < Base
# Initial stream, to be used for container creation (table usually)
# @param id string
# @return String
def initial_stream(id)
raise 'To be implemented in child classes'
end
# @param id string
# @return String
def stream_resource(id)
raise 'To be implemented in child classes'
end
private_class_method :new
end
end
end
@@ -0,0 +1,17 @@
module CartoDB
module Datasources
# Performs streaming from the datasource to a file
class BaseFileStream < Base
# @param id string
# @param stream Stream
# @return Integer bytes streamed
def stream_resource(id, stream)
raise 'To be implemented in child classes'
end
private_class_method :new
end
end
end
@@ -0,0 +1,93 @@
require_relative '../../../importer/lib/importer/unp'
module CartoDB
module Datasources
class BaseOAuth < Base
# At least for CartoDB SaaS, this will be parsed by oauth endpoint to rewrite the url, so must be filled with
# e.g. CALLBACK_STATE_DATA_PLACEHOLDER.sub('user', @user.username).sub('service', DATASOURCE_NAME)
# And appended anywhere in the querystring used as callback url with param "state=xxxxxx"
CALLBACK_STATE_DATA_PLACEHOLDER = '__user__service__'
attr_reader :config
def initialize(config, user, mandatory_config_parameters, datasource_name)
super
mandatory_config_parameters.each { |param| check_config(config, param, datasource_name) }
@config = config
@user = user
end
# TODO: Helper method to aid with migration of endpoints, can be removed after full AR migration
def service_name_for_user(service_name, user)
service_name
end
# Return the url to be displayed or sent the user to to authenticate and get authorization code
# @param use_callback_flow : bool
def get_auth_url(use_callback_flow=true)
raise 'To be implemented in child classes'
end #get_auth_url
# Validate authorization code and store token
# @param auth_code : string
# @return string : Access token
def validate_auth_code(auth_code)
raise 'To be implemented in child classes'
end #validate_auth_code
# Validates the authorization callback
# @param params : mixed
def validate_callback(params)
raise 'To be implemented in child classes'
end #validate_callback
# Set token
# @param token string
def token=(token)
raise 'To be implemented in child classes'
end #token=
# Retrieve token
# @return string | nil
def token
raise 'To be implemented in child classes'
end #token
# Checks if token is still valid or has been revoked
# @return bool
def token_valid?
raise 'To be implemented in child classes'
end #token_valid?
# Revokes current set token
def revoke_token
raise 'To be implemented in child classes'
end #revoke_token
private_class_method :new
protected
# Calculates a checksum of given input
# @param origin string
# @return string
def checksum_of(origin)
#noinspection RubyArgCount
Zlib::crc32(origin).to_s
end
def supported_extensions
CartoDB::Importer2::Unp::SUPPORTED_FORMATS
.concat(CartoDB::Importer2::Unp::COMPRESSED_EXTENSIONS)
end
private
def check_config(config, param, datasource_name)
raise MissingConfigurationError.new("missing #{param}", datasource_name) unless config.include?(param)
end
end
end
end
@@ -0,0 +1,147 @@
require_relative './url/arcgis'
require_relative './url/dropbox'
require_relative './url/box'
require_relative './url/gdrive'
require_relative './url/instagram_oauth'
require_relative './url/mailchimp'
require_relative './url/public_url'
require_relative 'search/twitter'
module CartoDB
module Datasources
class DatasourcesFactory
NAME = 'DatasourcesFactory'.freeze
# in seconds
HTTP_CONNECT_TIMEOUT = 60
DEFAULT_HTTP_REQUEST_TIMEOUT = 600
# Retrieve a datasource instance
# @param datasource_name string
# @param user ::User
# @param additional_config Hash
# {
# :redis_storage => Redis|nil
# :ogr2ogr_instance => Ogr2ogr|nil
# }
# @return mixed
# @throws MissingConfigurationError
def self.get_datasource(datasource_name, user, additional_config = {})
if additional_config[:http_timeout].nil?
additional_config[:http_timeout] = DEFAULT_HTTP_REQUEST_TIMEOUT
end
if additional_config[:http_connect_timeout].nil?
additional_config[:http_connect_timeout] = HTTP_CONNECT_TIMEOUT
end
case datasource_name
when Url::Dropbox::DATASOURCE_NAME
Url::Dropbox.get_new(DatasourcesFactory.config_for(datasource_name, user), user)
when Url::Box::DATASOURCE_NAME
Url::Box.get_new(DatasourcesFactory.config_for(datasource_name, user), user)
when Url::GDrive::DATASOURCE_NAME
Url::GDrive.get_new(DatasourcesFactory.config_for(datasource_name, user), user)
when Url::InstagramOAuth::DATASOURCE_NAME
Url::InstagramOAuth.get_new(DatasourcesFactory.config_for(datasource_name, user), user)
when Url::PublicUrl::DATASOURCE_NAME
Url::PublicUrl.get_new(additional_config)
when Url::ArcGIS::DATASOURCE_NAME
Url::ArcGIS.get_new(user)
when Url::MailChimp::DATASOURCE_NAME
Url::MailChimp.get_new(DatasourcesFactory.config_for(datasource_name, user).merge(additional_config), user)
when Search::Twitter::DATASOURCE_NAME
Search::Twitter.get_new(DatasourcesFactory.config_for(datasource_name, user), user,
additional_config[:redis_storage], additional_config[:user_defined_limits])
when nil
nil
else
raise MissingConfigurationError.new("unrecognized datasource #{datasource_name}", NAME)
end
end
# Gets all available oauth datasources
def self.get_all_oauth_datasources
[
Url::Dropbox::DATASOURCE_NAME,
Url::Box::DATASOURCE_NAME,
Url::GDrive::DATASOURCE_NAME,
# Url::InstagramOAuth::DATASOURCE_NAME,
Url::MailChimp::DATASOURCE_NAME
]
end
# Gets the config of a certain datasource
# @param datasource_name string
# @param user ::User
# @return string
# @throws MissingConfigurationError
def self.config_for(datasource_name, user)
config, datasource_supports_custom_config = get_config(datasource_name)
if datasource_supports_custom_config
key = customized_config_key(config, datasource_name, user)
if key.nil?
config[datasource_name][:standard.to_s]
else
# This code assumes config is ok
name_config_map = config[datasource_name]['entity_to_config_map'].select { |u| !u[key].nil? }.first
config[datasource_name][:customized.to_s][name_config_map[key]]
end
else
config.fetch(datasource_name)
end
end
def self.customized_config?(datasource_name, user)
config, datasource_supports_custom_config = get_config(datasource_name)
datasource_supports_custom_config && customized_config_key(config, datasource_name, user).present?
end
# Allows to set a custom config (useful for testing)
# @param custom_config string
def self.set_config(custom_config)
@forced_config = custom_config
end
def self.get_config(datasource_name)
config_source = @forced_config ? @forced_config : Cartodb.config
datasource_supports_custom_config = false
case datasource_name
when Url::Dropbox::DATASOURCE_NAME, Url::Box::DATASOURCE_NAME, Url::GDrive::DATASOURCE_NAME, Url::InstagramOAuth::DATASOURCE_NAME,
Url::MailChimp::DATASOURCE_NAME
config = (config_source[:oauth] rescue nil)
config ||= (config_source[:oauth.to_s] rescue nil)
when Search::Twitter::DATASOURCE_NAME
config = (config_source[:datasource_search] rescue nil)
config ||= (config_source[:datasource_search.to_s] rescue nil)
datasource_supports_custom_config = true
else
config = nil
end
if config.nil? || config.empty?
raise MissingConfigurationError.new("missing configuration for datasource #{datasource_name}", NAME)
end
[config, datasource_supports_custom_config]
end
private_class_method :get_config
def self.customized_config_key(config, datasource_name, user)
custom_config_orgs = config[datasource_name].fetch(:customized_orgs_list.to_s, [])
custom_config_users = config[datasource_name][:customized_user_list.to_s]
if user.organization_user? && custom_config_orgs.include?(user.organization.name)
user.organization.name
elsif custom_config_users.include?(user.username)
user.username
end
end
private_class_method :customized_config_key
end
end
end
@@ -0,0 +1,25 @@
module CartoDB
module Datasources
module Decorators
class BaseDecorator
# @return bool
def decorates_layer?
raise 'To be implemented in child classes'
end
# @param layer Layer|nil
# @return bool
def layer_eligible?(layer=nil)
raise 'To be implemented in child classes'
end
# @param layer Layer|nil
def decorate_layer!(layer=nil)
raise 'To be implemented in child classes'
end
end
end
end
end
@@ -0,0 +1,27 @@
require_relative '../url/instagram_oauth'
require_relative './base_decorator'
require_relative './instagram_decorator'
require_relative './mailchimp_decorator'
module CartoDB
module Datasources
module Decorators
class Factory
def self.decorator_for(data_import_service_name='')
case data_import_service_name
when Url::InstagramOAuth::DATASOURCE_NAME
Decorators::InstagramDecorator.new
when Url::MailChimp::DATASOURCE_NAME
Decorators::MailchimpDecorator.new
else
nil
end
end
end
end
end
end
@@ -0,0 +1,46 @@
require_relative './base_decorator'
module CartoDB
module Datasources
module Decorators
class InstagramDecorator < BaseDecorator
# @return bool
def decorates_layer?
true
end
# @param layer Layer|nil
# @return bool
def layer_eligible?(layer=nil)
return false if layer.nil?
# Only data/cartodb layers
return false unless layer.respond_to?(:data_layer?)
layer.data_layer?
end
# @param layer Layer|nil
def decorate_layer!(layer=nil)
return nil unless layer_eligible?(layer)
layer.infowindow = {
fields: [
{position: 0, name: "thumbnail", title: true},
{position: 1, name: "caption", title: true},
{position: 2, name: "comments_count", title: true},
{position: 3, name: "likes_count", title: true},
{position: 4, name: "link", title: true}
],
template_name: "infowindow_header_with_image",
alternative_names: {},
maxHeight: 275,
width: 226,
template: ""
}
nil
end
end
end
end
end
@@ -0,0 +1,135 @@
require_relative './base_decorator'
module CartoDB
module Datasources
module Decorators
class MailchimpDecorator < BaseDecorator
CATEGORY_COLUMN = 'opened'
CSS_PROPERTIES = {
"marker-opacity" => 1,
"marker-fill-opacity" => 0.5,
"marker-line-color" => "#FFF",
"marker-line-width" => 1,
"marker-line-opacity" => 1,
"marker-placement" => "point",
"marker-type" => "ellipse",
"marker-width" => 6,
"marker-allow-overlap" => true,
"marker-comp-op" => 'multiply'
}
CATEGORIES = [
{
title: true,
color: "#A53ED5"
},
{
title: false,
color: "#00ceff"
}
]
# @return bool
def decorates_layer?
true
end
# @param layer Layer|nil
# @return bool
def layer_eligible?(layer=nil)
return false if layer.nil?
# Only data/cartodb layers
return false unless layer.respond_to?(:data_layer?)
layer.data_layer?
end
# @param layer Layer|nil
def decorate_layer!(layer=nil)
return nil unless layer_eligible?(layer)
enable_category_wizard(layer)
enable_category_legend(layer)
set_carto_css(layer)
nil
end
private
def enable_category_wizard(layer)
wizard_properties = {
type: "category",
properties: {
property: CATEGORY_COLUMN,
"geometry_type" => "point",
categories: []
}
}
wizard_properties[:properties].merge!(CSS_PROPERTIES)
wizard_properties[:properties][:categories] = CATEGORIES.map do |category|
{
"title" => category[:title],
"title_type" => "boolean",
"color" => category[:color],
"value_type" => "color"
}
end
layer.set_option('wizard_properties', wizard_properties)
end
def enable_category_legend(layer)
legend = {
"type" => "category",
"show_title" => false,
"title" => "",
"template" => "",
"visible" => true
}
legend[:items] = CATEGORIES.map do |category|
{
name: category[:title].to_s,
visible: true,
value: category[:color]
}
end
layer.set_option(:legend, legend)
end
def set_carto_css(layer)
matches = layer.options['tile_style'].match(/^#(.*) \{/)
unless matches.nil?
css_selector = "##{matches[1]}"
css_properties = CSS_PROPERTIES.map{|property, value| " #{property}: #{value};"}.join("\n")
carto_css = []
carto_css << "#{css_selector} {"
carto_css << css_properties
carto_css << " [zoom>4] {"
carto_css << " marker-width: 7;"
carto_css << " }"
carto_css << " [zoom>5] {"
carto_css << " marker-width: 8;"
carto_css << " }"
carto_css << " [zoom>6] {"
carto_css << " marker-width: 9;"
carto_css << " }"
carto_css << "}"
CATEGORIES.each do |category|
carto_css << "#{css_selector}[#{CATEGORY_COLUMN}=#{category[:title]}] {"
carto_css << " marker-fill: #{category[:color]};"
carto_css << "}"
end
layer.set_option('tile_style', carto_css.join("\n"))
layer.set_option('tile_style_custom', false)
end
end
end
end
end
end
@@ -0,0 +1,66 @@
module CartoDB
module Datasources
# Remember to add new errors to:
# config/initializers/carto_db.rb
# services/importer/lib/importer/exceptions.rb
class DatasourceBaseError < StandardError
UNKNOWN_SERVICE = 'UNKNOWN'.freeze
attr_reader :service_name
def initialize(message = 'General error', service = UNKNOWN_SERVICE, username = nil)
@service_name = service
message = "#{message}"
message << " @ #{@service_name}" if @service_name != UNKNOWN_SERVICE
message << " User: #{username}" unless username.nil?
super(message)
end
end
class AuthError < DatasourceBaseError; end
# This exception is ONLY throwed if oauth token is wrong or expired, and should be deleted if exists
class TokenExpiredOrInvalidError < AuthError; end
class InvalidServiceError < DatasourceBaseError; end
class DataDownloadError < DatasourceBaseError; end
class UnsupportedOperationError < DatasourceBaseError; end
class NotFoundDownloadError < DatasourceBaseError; end
class MissingConfigurationError < DatasourceBaseError; end
class UninitializedError < DatasourceBaseError; end
class NoResultsError < DatasourceBaseError; end
class ParameterError < DatasourceBaseError; end
class OutOfQuotaError < DatasourceBaseError; end
class InvalidInputDataError < DatasourceBaseError; end
class ResponseError < DatasourceBaseError; end
class ExternalServiceError < DatasourceBaseError; end
class GNIPServiceError < ExternalServiceError; end
class ServiceDisabledError < DatasourceBaseError
def initialize(service = UNKNOWN_SERVICE, username = nil)
super("Service disabled", service, username)
end
end
class DataDownloadTimeoutError < DatasourceBaseError
def initialize(service = UNKNOWN_SERVICE, username = nil)
super("Data download timed out. Check the source is not running slow and/or try again.", service, username)
end
end
class ExternalServiceTimeoutError < DatasourceBaseError
def initialize(service = UNKNOWN_SERVICE, username = nil)
super("External service timed out. Check the source is not running slow and/or try again.", service, username)
end
end
class DatasourcePermissionError < DatasourceBaseError; end
class DropboxPermissionError < DatasourcePermissionError; end
class BoxPermissionError < DatasourcePermissionError; end
class GDriveNoExternalAppsAllowedError < DatasourceBaseError; end
end
end
@@ -0,0 +1,597 @@
require 'json'
require_relative '../util/csv_file_dumper'
require_relative '../../../../twitter-search/twitter-search'
require_relative '../../../../../lib/cartodb/logger'
require_relative '../base_file_stream'
module CartoDB
module Datasources
module Search
# NOTE: 'redis_storage' is only sent in normal imports, not at OAuth or Synchronizations,
# as this datasource is not intended to be used in such.
class Twitter < BaseFileStream
# Required for all datasources
DATASOURCE_NAME = 'twitter_search'
NO_TOTAL_RESULTS = -1
MAX_CATEGORIES = 4
DEBUG_FLAG = false
# Used for each query page size, not as total
FILTER_MAXRESULTS = :maxResults
FILTER_FROMDATE = :fromDate
FILTER_TODATE = :toDate
FILTER_CATEGORIES = :categories
FILTER_TOTAL_RESULTS = :totalResults
USER_LIMITS_FILTER_CREDITS = :twitter_credits_limit
CATEGORY_NAME_KEY = :name
CATEGORY_TERMS_KEY = :terms
GEO_SEARCH_FILTER = 'has:geo'
PROFILE_GEO_SEARCH_FILTER = 'has:profile_geo'
OR_SEARCH_FILTER = 'OR'
# Seconds to substract from current time as threshold to consider a time
# as "now or from the future" upon date filter build
TIMEZONE_THRESHOLD = 60
# Gnip's 30 limit minus 'has:geo' one
MAX_SEARCH_TERMS = 30 - 1
MAX_QUERY_SIZE = 2048
MAX_TABLE_NAME_SIZE = 30
# Constructor
# @param config Array
# [
# 'auth_required'
# 'username'
# 'password'
# 'search_url'
# ]
# @param user ::User
# @param redis_storage Redis|nil (optional)
# @param user_defined_limits Hash|nil (optional)
# @throws UninitializedError
def initialize(config, user, redis_storage = nil, user_defined_limits={})
@service_name = DATASOURCE_NAME
@filters = Hash.new
raise UninitializedError.new('missing user instance', DATASOURCE_NAME) if user.nil?
raise MissingConfigurationError.new('missing auth_required', DATASOURCE_NAME) unless config.include?('auth_required')
raise MissingConfigurationError.new('missing username', DATASOURCE_NAME) unless config.include?('username')
raise MissingConfigurationError.new('missing password', DATASOURCE_NAME) unless config.include?('password')
raise MissingConfigurationError.new('missing search_url for GNIP API', DATASOURCE_NAME) unless config.include?('search_url')
@user_defined_limits = user_defined_limits
@search_api_config = {
TwitterSearch::SearchAPI::CONFIG_AUTH_REQUIRED => config['auth_required'],
TwitterSearch::SearchAPI::CONFIG_AUTH_USERNAME => config['username'],
TwitterSearch::SearchAPI::CONFIG_AUTH_PASSWORD => config['password'],
TwitterSearch::SearchAPI::CONFIG_SEARCH_URL => config['search_url'],
TwitterSearch::SearchAPI::CONFIG_REDIS_RL_ACTIVE => config.fetch('ratelimit_active', nil),
TwitterSearch::SearchAPI::CONFIG_REDIS_RL_MAX_CONCURRENCY => config.fetch('ratelimit_concurrency', nil),
TwitterSearch::SearchAPI::CONFIG_REDIS_RL_TTL => config.fetch('ratelimit_ttl', nil),
TwitterSearch::SearchAPI::CONFIG_REDIS_RL_WAIT_SECS => config.fetch('ratelimit_wait_secs', nil)
}
@redis_storage = redis_storage
@csv_dumper = CSVFileDumper.new(TwitterSearch::JSONToCSVConverter.new, DEBUG_FLAG)
@user = user
@data_import_item = nil
@logger = nil
@used_quota = 0
@user_semaphore = Mutex.new
end
# Factory method
# @param config {}
# @param user ::User
# @param redis_storage Redis|nil
# @param user_defined_limits Hash|nil
# @return CartoDB::Datasources::Search::TwitterSearch
def self.get_new(config, user, redis_storage = nil, user_defined_limits={})
return new(config, user, redis_storage, user_defined_limits)
end
# If will provide a url to download the resource, or requires calling get_resource()
# @return bool
def providers_download_url?
false
end
# Perform the listing and return results
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
# @return [ { :id, :title, :url, :service } ]
def get_resources_list(filter=[])
filter
end
# @param id string
# @param stream Stream
# @return Integer bytes streamed
def stream_resource(id, stream)
unless has_enough_quota?(@user)
raise OutOfQuotaError.new("#{@user.username} out of quota for tweets", DATASOURCE_NAME)
end
raise ServiceDisabledError.new(DATASOURCE_NAME, @user.username) unless is_service_enabled?(@user)
fields_from(id)
do_search(@search_api_config, @redis_storage, @filters, stream)
end
# Retrieves a resource and returns its contents
# @param id string Will contain a stringified JSON
# @return mixed
# @throws ServiceDisabledError
# @throws OutOfQuotaError
# @throws ParameterError
# @deprecated Use stream_resource instead
def get_resource(id)
unless has_enough_quota?(@user)
raise OutOfQuotaError.new("#{@user.username} out of quota for tweets", DATASOURCE_NAME)
end
raise ServiceDisabledError.new(DATASOURCE_NAME, @user.username) unless is_service_enabled?(@user)
fields_from(id)
do_search(@search_api_config, @redis_storage, @filters, stream = nil)
end
# @param id string
# @return Hash
def get_resource_metadata(id)
fields_from(id)
{
id: id,
title: DATASOURCE_NAME,
url: nil,
service: DATASOURCE_NAME,
checksum: nil,
size: 0,
filename: "#{table_name}.csv"
}
end
# Retrieves current filters. Unused as here there's no get_resources_list
# @return {}
def filter
{}
end
# Sets current filters. Unused as here there's no get_resources_list
# @param filter_data {}
def filter=(filter_data=[])
filter_data
end
# Hide sensitive fields
def to_s
"<CartoDB::Datasources::Search::Twitter @user=#{@user.username} @filters=#{@filters} @search_api_config=#{search_api_config_public_values}>"
end
# If this datasource accepts a data import instance
# @return Boolean
def persists_state_via_data_import?
true
end
# Stores the data import item instance to use/manipulate it
# @param value DataImport
def data_import_item=(value)
@data_import_item = value
end
def set_audit_to_completed(table_id = nil)
entry = audit_entry.class.where(data_import_id:@data_import_item.id).first
raise DatasourceBaseError.new("Couldn't fetch SearchTweet entry for data import #{@data_import_item.id}", \
DATASOURCE_NAME) if entry.nil?
entry.set_complete_state
entry.table_id = table_id unless table_id.nil?
entry.save
end
def set_audit_to_failed
entry = audit_entry.class.where(data_import_id:@data_import_item.id).first
raise DatasourceBaseError.new("Couldn't fetch SearchTweet entry for data import #{@data_import_item.id}", \
DATASOURCE_NAME) if entry.nil?
entry.set_failed_state
entry.save
end
# @return Hash
def get_audit_stats
entry = audit_entry.class.where(data_import_id:@data_import_item.id).first
raise DatasourceBaseError.new("Couldn't fetch SearchTweet entry for data import #{@data_import_item.id}", \
DATASOURCE_NAME) if entry.nil?
{ :retrieved_items => entry.retrieved_items }
end
private
# Used at specs
attr_accessor :search_api_config, :csv_dumper
attr_reader :data_import_item
def search_api_config_public_values
{
TwitterSearch::SearchAPI::CONFIG_AUTH_REQUIRED =>
@search_api_config[TwitterSearch::SearchAPI::CONFIG_AUTH_REQUIRED],
TwitterSearch::SearchAPI::CONFIG_AUTH_USERNAME =>
@search_api_config[TwitterSearch::SearchAPI::CONFIG_AUTH_USERNAME],
TwitterSearch::SearchAPI::CONFIG_SEARCH_URL =>
@search_api_config[TwitterSearch::SearchAPI::CONFIG_SEARCH_URL]
}
end
# Returns if the user set a maximum credits to use
# @return Integer
def twitter_credit_limits
@user_defined_limits.fetch(USER_LIMITS_FILTER_CREDITS, 0)
end
# Wraps check of specified user limit or not (to use instead his max quota)
# @return Integer
def remaining_quota
twitter_credit_limits > 0 ? [@user.remaining_twitter_quota, twitter_credit_limits].min
: @user.remaining_twitter_quota
end
def table_name
terms_fragment = @filters[FILTER_CATEGORIES].map { |category|
clean_category(category[CATEGORY_TERMS_KEY]).gsub(/[^0-9a-z,]/i, '').gsub(/[,]/i, '_')
}.join('_').slice(0,MAX_TABLE_NAME_SIZE)
"twitter_#{terms_fragment}"
end
def clean_category(category)
category.gsub(" (#{GEO_SEARCH_FILTER} OR #{PROFILE_GEO_SEARCH_FILTER})", '')
.gsub(" #{OR_SEARCH_FILTER} ", ', ')
.gsub(/^\(/, '')
.gsub(/\)$/, '')
end
def fields_from(id)
return unless @filters.count == 0
fields = ::JSON.parse(id, symbolize_names: true)
@filters[FILTER_CATEGORIES] = build_queries_from_fields(fields)
if @filters[FILTER_CATEGORIES].size > MAX_CATEGORIES
raise ParameterError.new("Max allowed categories are #{FILTER_CATEGORIES}", DATASOURCE_NAME)
end
@filters[FILTER_FROMDATE] = build_date_from_fields(fields, 'from')
@filters[FILTER_TODATE] = build_date_from_fields(fields, 'to')
@filters[FILTER_MAXRESULTS] = build_maxresults_field(@user)
@filters[FILTER_TOTAL_RESULTS] = build_total_results_field(@user)
end
# Signature must be like: .report_message('Import error', 'error', error_info: stacktrace)
def report_error(message, additional_data)
log("Error: #{message} Additional Info: #{additional_data}")
CartoDB::Logger.error(message: message, error_info: additional_data)
end
# @param api_config Hash
# @param redis_storage Mixed
# @param filters Hash
# @param stream IO
# @return Mixed The data
def do_search(api_config, redis_storage, filters, stream)
threads = {}
base_filters = filters.select { |k, v| k != FILTER_CATEGORIES }
category_totals = {}
dumper_additional_fields = {}
filters[FILTER_CATEGORIES].each { |category|
dumper_additional_fields[category[CATEGORY_NAME_KEY]] = {
category_name: category[CATEGORY_NAME_KEY],
category_terms: clean_category(category[CATEGORY_TERMS_KEY])
}
@csv_dumper.begin_dump(category[CATEGORY_NAME_KEY])
}
@csv_dumper.additional_fields = dumper_additional_fields
log("Searching #{filters[FILTER_CATEGORIES].length} categories")
filters[FILTER_CATEGORIES].each { |category|
# If all threads are created at the same time, redis semaphore inside search_api
# might not yet have new value, so introduce a small delay on each thread creation
sleep(0.1)
threads[category[CATEGORY_NAME_KEY]] = Thread.new {
api = TwitterSearch::SearchAPI.new(api_config, redis_storage, @csv_dumper)
# Dumps happen inside upon each block response
total_results = search_by_category(api, base_filters, category)
category_totals[category[CATEGORY_NAME_KEY]] = total_results
}
}
threads.each {|key, thread|
thread.join
}
# INFO: For now we don't treat as error a no results scenario, else use:
# raise NoResultsError.new if category_totals.values.inject(:+) == 0
filters[FILTER_CATEGORIES].each { |category|
@csv_dumper.end_dump(category[CATEGORY_NAME_KEY])
}
streamed_size = @csv_dumper.merge_dumps_into_stream(dumper_additional_fields.keys, stream)
log("Temp files:\n#{@csv_dumper.file_paths}")
log("#{@csv_dumper.original_file_paths}\n#{@csv_dumper.headers_path}")
if twitter_credit_limits > 0 || !@user.soft_twitter_datasource_limit
if (remaining_quota - @used_quota) < 0
# Make sure we don't charge extra tweets (even if we "lose" charging a block or two of tweets)
@used_quota = remaining_quota
end
end
# remaining quota is calc. on the fly based on audits/imports
save_audit(@user, @data_import_item, @used_quota)
streamed_size
end
def search_by_category(api, base_filters, category)
api.params = base_filters
exception = nil
next_results_cursor = nil
total_results = 0
begin
exception = nil
out_of_quota = false
@user_semaphore.synchronize {
# Credit limits must be honoured above soft limit
if twitter_credit_limits > 0 || !@user.soft_twitter_datasource_limit
if remaining_quota - @used_quota <= 0
out_of_quota = true
next_results_cursor = nil
end
end
}
unless out_of_quota
api.query_param = category[CATEGORY_TERMS_KEY]
begin
results_page = api.fetch_results(next_results_cursor)
rescue TwitterSearch::TwitterHTTPException => e
exception = e
report_error(e.to_s, e.backtrace)
# Stop gracefully to not break whole import process
results_page = {
results: [],
next: nil
}
end
dumped_items_count = @csv_dumper.dump(category[CATEGORY_NAME_KEY], results_page[:results])
next_results_cursor = results_page[:next].nil? ? nil : results_page[:next]
@user_semaphore.synchronize {
@used_quota += dumped_items_count
}
total_results += dumped_items_count
end
end while (!next_results_cursor.nil? && !out_of_quota && !exception)
log("'#{category[CATEGORY_NAME_KEY]}' got #{total_results} results")
log("Got exception at '#{category[CATEGORY_NAME_KEY]}': #{exception.inspect}") if exception
# If fails on the first request, bubble up the error, else will return as many tweets as possible
if !exception.nil? && total_results == 0
log("ERROR: 0 results & exception: #{exception} (HTTP #{exception.http_code}) #{exception.additional_data}")
# @see http://support.gnip.com/apis/search_api/api_reference.html
if exception.http_code == 422 && exception.additional_data =~ /request usage cap exceeded/i
raise OutOfQuotaError.new(exception.to_s, DATASOURCE_NAME)
end
if [401, 404].include?(exception.http_code)
raise MissingConfigurationError.new(exception.to_s, DATASOURCE_NAME)
end
if [400, 422].include?(exception.http_code)
raise InvalidInputDataError.new(exception.to_s, DATASOURCE_NAME)
end
if exception.http_code == 429
raise ResponseError.new(exception.to_s, DATASOURCE_NAME)
end
if exception.http_code >= 500 && exception.http_code < 600
raise GNIPServiceError.new(exception.to_s, DATASOURCE_NAME)
end
raise DatasourceBaseError.new(exception.to_s, DATASOURCE_NAME)
end
total_results
end
def build_date_from_fields(fields, date_type)
raise ParameterError.new('missing dates', DATASOURCE_NAME) \
if fields[:dates].nil?
case date_type
when 'from'
date_sym = :fromDate
hour_sym = :fromHour
min_sym = :fromMin
when 'to'
date_sym = :toDate
hour_sym = :toHour
min_sym = :toMin
else
raise ParameterError.new("unknown date type #{date_type}", DATASOURCE_NAME)
end
if fields[:dates][date_sym].nil? || fields[:dates][hour_sym].nil? || fields[:dates][min_sym].nil?
date = nil
else
# Sent by JS in minutes
timezone = fields[:dates][:user_timezone].nil? ? 0 : fields[:dates][:user_timezone].to_i
begin
year, month, day = fields[:dates][date_sym].split('-')
timezoned_date = Time.gm(year, month, day, fields[:dates][hour_sym], fields[:dates][min_sym])
rescue ArgumentError
raise ParameterError.new('Invalid date format', DATASOURCE_NAME)
end
timezoned_date += timezone*60
# Gnip doesn't allows searches "in the future"
date = timezoned_date >= (Time.now - TIMEZONE_THRESHOLD).utc ? nil : timezoned_date.strftime("%Y%m%d%H%M")
end
date
end
def build_queries_from_fields(fields)
raise ParameterError.new('missing categories', DATASOURCE_NAME) \
if fields[:categories].nil? || fields[:categories].empty?
queries = []
fields[:categories].each { |category|
raise ParameterError.new('missing category', DATASOURCE_NAME) if category[:category].nil?
raise ParameterError.new('missing terms', DATASOURCE_NAME) if category[:terms].nil?
# Gnip limitation
if category[:terms].count > MAX_SEARCH_TERMS
category[:terms] = category[:terms].slice(0, MAX_SEARCH_TERMS)
end
category[:terms] = sanitize_terms(category[:terms])
query = {
CATEGORY_NAME_KEY => category[:category].to_s,
CATEGORY_TERMS_KEY => ''
}
unless category[:terms].count == 0
query[CATEGORY_TERMS_KEY] << '('
query[CATEGORY_TERMS_KEY] << category[:terms].join(' OR ')
query[CATEGORY_TERMS_KEY] << ") (#{GEO_SEARCH_FILTER} OR #{PROFILE_GEO_SEARCH_FILTER})"
end
if query[CATEGORY_TERMS_KEY].length > MAX_QUERY_SIZE
raise ParameterError.new("Obtained search query is bigger than #{MAX_QUERY_SIZE} chars", DATASOURCE_NAME)
end
queries << query
}
queries
end
# @param terms_list Array
def sanitize_terms(terms_list)
terms_list.map{ |term|
# Remove unwanted stuff
sanitized = term.to_s.gsub(/^ /, '').gsub(/ $/, '').gsub('"', '')
# Quote if needed
if sanitized.gsub(/[a-z0-9@#]/i,'') != ''
sanitized = '"' + sanitized + '"'
end
sanitized.length == 0 ? nil : sanitized
}.compact
end
# Max results per page
# @param user ::User
def build_maxresults_field(user)
if twitter_credit_limits > 0
[remaining_quota, TwitterSearch::SearchAPI::MAX_PAGE_RESULTS].min
else
# user about to hit quota?
if remaining_quota < TwitterSearch::SearchAPI::MAX_PAGE_RESULTS
if user.soft_twitter_datasource_limit
# But can go beyond limits
TwitterSearch::SearchAPI::MAX_PAGE_RESULTS
else
remaining_quota
end
else
TwitterSearch::SearchAPI::MAX_PAGE_RESULTS
end
end
end
# Max total results
# @param user ::User
def build_total_results_field(user)
if twitter_credit_limits == 0 && user.soft_twitter_datasource_limit
NO_TOTAL_RESULTS
else
remaining_quota
end
end
# @param user ::User
def is_service_enabled?(user)
if !user.organization.nil?
enabled = user.organization.twitter_datasource_enabled
if enabled
user.twitter_datasource_enabled
else
# If disabled org-wide, disabled for everyone
false
end
else
user.twitter_datasource_enabled
end
end
# @param user ::User
# @return boolean
def has_enough_quota?(user)
# As this is used to disallow searches (and throw exceptions) don't use here user limits
user.soft_twitter_datasource_limit || (user.remaining_twitter_quota > 0)
end
# @param user ::User
# @param data_import_item DataImport
# @param retrieved_items_count Integer
def save_audit(user, data_import_item, retrieved_items_count)
entry = audit_entry
entry.set_importing_state
entry.user_id = user.id
entry.data_import_id = data_import_item.id
entry.service_item_id = data_import_item.service_item_id
entry.retrieved_items = retrieved_items_count
entry.save
end
# Call this inside specs to override returned class
# @param override_class SearchTweet|nil (optional)
# @return SearchTweet
def audit_entry(override_class = nil)
if @audit_entry.nil?
if override_class.nil?
require_relative '../../../../../app/models/search_tweet'
@audit_entry = ::SearchTweet.new
else
@audit_entry = override_class.new
end
end
@audit_entry
end
end
end
end
end
@@ -0,0 +1,540 @@
require 'json'
require 'addressable/uri'
require_relative '../base_direct_stream'
require_relative '../../../../../lib/carto/http/client'
module CartoDB
module Datasources
module Url
class ArcGIS < BaseDirectStream
# Required for all datasources
DATASOURCE_NAME = 'arcgis'
ARCGIS_API_LIKE_URL_RE = /\/rest\/services/i
METADATA_URL = '%s?f=json'
FEATURE_IDS_URL = '%s/query?where=1%%3D1&returnIdsOnly=true&f=json'
FEATURE_DATA_POST_URL = '%s/query'
LAYERS_URL = '%s/layers?f=json'
MINIMUM_SUPPORTED_VERSION = 10.1
# In seconds, for connecting
HTTP_CONNECTION_TIMEOUT = 60
# In seconds, for the full request
HTTP_TIMEOUT = 60
# Amount to multiply or divide
BLOCK_FACTOR = 2
MIN_BLOCK_SIZE = 1
# GeoJSON can get too big in memory, or ArcGIS have mem problems, so keep reasonable number
MAX_BLOCK_SIZE = 100
# In seconds, use 0 to disable
BLOCK_SLEEP_TIME = 1
# Each retry will be after SLEEP_REQUEST_TIME^(current_retries_count). Set to 0 to disable retrying
MAX_RETRIES = 0
SLEEP_REQUEST_TIME = 3
SKIP_FAILED_IDS = true
# Used to display more data only (for local debugging purposes)
DEBUG = false
VECTOR_LAYER_TYPE = 'Feature Layer'.freeze
OID_FIELD_TYPE = 'esriFieldTypeOID'.freeze
attr_reader :metadata
# Constructor
# @param user ::User
def initialize(user)
super
@service_name = DATASOURCE_NAME
# Fields:
# @metadata = {
# arcgis_version: nil,
# name: nil,
# description: nil,
# type: nil,
# geometry_type: nil,
# copyright: nil,
# fields: [],
# max_records_per_query: 500,
# supported_formats: [],
# advanced_queries_supported: false
# }
@metadata = nil
@user = user
@url = nil
@ids_total = 0
@ids_retrieved = 0
@block_size = 0
@current_stream_status = true
@last_stream_status = true
@ids = nil
end
# Factory method
# @param user ::User
# @return CartoDB::Datasources::Url::ArcGIS
def self.get_new(user)
return new(user)
end
# @return String
def to_s
"<CartoDB::Datasources::Url::ArcGis @url=#{@url} @metadata=#{@metadata} @ids_total=#{@ids_total}" +
" @ids_retrieved=#{@ids_retrieved} current_block_size=#{block_size(update=false)}>"
end
# If will provide a url to download the resource, or requires calling get_resource()
# @return Bool
def providers_download_url?
false
end
# Perform the listing and return results
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
# @return [ Hash ]
def get_resources_list(filter=[])
filter
end
# Retrieves a resource and returns its contents
# @param id string
# @return mixed
def get_resource(id)
raise 'Not supported by this datasource'
end
# Initial stream, to be used for container creation (table usually)
# @param id string
# @return String
def initial_stream(id)
sub_id = get_subresource_id(id)
@url = sanitize_id(id, sub_id)
@ids = get_ids_list(@url)
@ids_total = @ids.length
first_item = get_by_ids(@url, [@ids.slice!(0)], @metadata[:fields])
@ids_retrieved += 1
# Start optimistic
@block_size = [MAX_BLOCK_SIZE, @metadata[:max_records_per_query]].min
::JSON.dump(first_item)
end
# @param id string
# @return String|nil Nil if no more items
def stream_resource(id)
return nil if @ids.empty?
retries = 0
begin
ids_block = @ids.slice!(0, [@ids.length, block_size].min)
puts "#{@ids_retrieved}/#{@ids_total} (#{ids_block.length})" if DEBUG
items = get_by_ids(@url, ids_block, @metadata[:fields])
@last_stream_status = @current_stream_status
@current_stream_status = true
retries = 0
sleep(BLOCK_SLEEP_TIME) unless BLOCK_SLEEP_TIME == 0
rescue ExternalServiceError => exception
if @block_size == MIN_BLOCK_SIZE && retries >= MAX_RETRIES
if SKIP_FAILED_IDS
items = []
else
raise exception
end
else
@last_stream_status = @current_stream_status
@current_stream_status = false
# Add back, next pass will get fewer items
@ids = ids_block + @ids
if @block_size == MIN_BLOCK_SIZE
retries += 1
sleep_time = SLEEP_REQUEST_TIME ** retries
puts "Retry delay (#{sleep_time}s)" if DEBUG
sleep(sleep_time)
end
retry
end
end
@ids_retrieved += ids_block.length
items.length > 0 ? ::JSON.dump(items) : ''
end
# @param id string
# @return Hash
# @throws DataDownloadError
# @throws ResponseError
# @throws InvalidServiceError
# @throws ServiceDisabledError
def get_resource_metadata(id)
if is_multiresource?(id)
@url = sanitize_id(id)
{
# Store original id, not the sanitized one
id: id,
subresources: get_layers_list(@url)
}
else
sub_id = get_subresource_id(id)
get_subresource_metadata(id, sub_id)
end
end
# Retrieves current filters. Unused as here there's no get_resources_list
# @return {}
def filter
{}
end
# Sets current filters. Unused as here there's no get_resources_list
# @param filter_data {}
def filter=(filter_data=[])
filter_data
end
# If this datasource accepts a data import instance
# @return Boolean
def persists_state_via_data_import?
false
end
# If true, a single resource id might return >1 subresources (each one spawning a table)
# @param id String
# @return Bool
def multi_resource_import_supported?(id)
is_multiresource?(id)
end
private
def http_client
@http_client ||= Carto::Http::Client.get('arcgis')
end
# @param id String
# @param subresource_id String
# @return Hash
# @throws DataDownloadError
# @throws ResponseError
# @throws InvalidServiceError
def get_subresource_metadata(id, subresource_id)
@url = sanitize_id(id, subresource_id)
response = http_client.get(METADATA_URL % [@url], http_options)
validate_response(METADATA_URL % [@url], response)
# non-rails symbolize keys
data = ::JSON.parse(response.body).inject({}){|memo,(k,v)| memo[k.to_sym] = v; memo}
raise ResponseError.new("Invalid layer type: '#{data[:type]}'") if data[:type] != VECTOR_LAYER_TYPE
raise ResponseError.new("Missing data: 'fields'") if data[:fields].nil?
if data[:supportedQueryFormats].present?
supported_formats = data.fetch(:supportedQueryFormats).gsub(' ', '').split(',')
else
supported_formats = []
end
begin
@metadata = {
arcgis_version: data.fetch(:currentVersion),
name: data.fetch(:name),
description: data.fetch(:description, ''),
type: data.fetch(:type),
geometry_type: data.fetch(:geometryType),
copyright: data.fetch(:copyrightText, ''),
fields: data.fetch(:fields).try(:map) { |field|
{
name: field['name'],
type: field['type']
}
},
max_records_per_query: data.fetch(:maxRecordCount, 500),
supported_formats: supported_formats,
advanced_queries_supported: data.fetch(:supportsAdvancedQueries, false)
}
rescue => exception
raise ResponseError.new("Missing data: #{exception.to_s} #{exception.backtrace}")
end
raise InvalidServiceError.new("Unsupported ArcGIS version #{@metadata[:arcgis_Version]}, must be >= #{MINIMUM_SUPPORTED_VERSION}") \
if @metadata[:arcgis_version] < MINIMUM_SUPPORTED_VERSION
{
id: id,
title: @metadata[:name],
url: nil,
service: DATASOURCE_NAME,
checksum: nil,
size: NO_CONTENT_SIZE_PROVIDED,
filename: filename_from(@metadata[:name])
}
end
# Just detects if id is a full map or a specific layer
# @param id String
# @return Bool
def is_multiresource?(id)
unless id.rindex('?').nil?
id = id.slice(0, id.rindex('?'))
end
(id =~ /\/([0-9]+\/|[0-9]+)$/).nil?
end
def get_subresource_id(id)
id.match(/([0-9]+\/|[0-9]+)$?/)[0]
end
# @param id String
# @param sub_id String|nil
# @return String
# @throws InvalidInputDataError
def sanitize_id(id, sub_id=nil)
# http://<host>/<site>/rest/services/<folder>/<serviceName>/<serviceType>/
# <site> is almost always "arcgis" (according to official doc)
unless id =~ ARCGIS_API_LIKE_URL_RE
raise InvalidInputDataError.new("Url doesn't looks as from ArcGIS server")
end
unless id.rindex('?').nil?
id = id.slice(0, id.rindex('?'))
end
if is_multiresource?(id) && !sub_id.nil?
id = (id.end_with?('/') ? id : id + '/') + sub_id
end
id
end
# @return Array [ { :id, :title} ]
# @throws DataDownloadError
# @throws ResponseError
def get_layers_list(url)
request_url = LAYERS_URL % [url]
response = http_client.get(request_url, http_options)
validate_response(request_url, response)
begin
data = ::JSON.parse(response.body).fetch('layers')
rescue => exception
raise ResponseError.new("Missing data: #{exception.to_s} #{request_url} #{exception.backtrace}")
end
# We only support vector layers (not raster layers)
data = data.reject { |layer| layer['type'] != VECTOR_LAYER_TYPE }
raise ResponseError.new("Empty layers list #{request_url}") if data.length == 0
begin
data.collect { |item|
{
# Leave prepared all child urls
id: (url.end_with?('/') ? url : url + '/') + item.fetch('id').to_s,
title: item.fetch('name')
}
}
rescue => exception
raise ResponseError.new("Missing data: #{exception.to_s} #{request_url} #{exception.backtrace}")
end
end
# NOTE: Assumes url is valid
# NOTE: Returned ids are sorted so they can be chunked into blocks to
# be requested by range queries: `(OBJECTID >= ... AND OBJECTID <= ... )`
# @param url String
# @return Array
# @throws DataDownloadError
# @throws ResponseError
def get_ids_list(url)
request_url = FEATURE_IDS_URL % [url]
response = http_client.get(request_url, http_options)
validate_response(request_url, response)
begin
data = ::JSON.parse(response.body).fetch('objectIds').sort
rescue => exception
raise ResponseError.new("Missing data: #{exception.to_s} #{request_url} #{exception.backtrace}")
end
raise ResponseError.new("Empty ids list #{request_url}") if data.length == 0
data
end
# NOTE: Assumes url is valid
# @param url String
# @param ids Array
# @param fields Array
# @return Array [ Hash ] (non-symbolized keys)
# @throws InvalidInputDataError
# @throws DataDownloadError
# @throws ExternalServiceError
def get_by_ids(url, ids, fields)
raise InvalidInputDataError.new("'ids' empty or invalid") if (ids.nil? || ids.length == 0)
raise InvalidInputDataError.new("'fields' empty or invalid") if (fields.nil? || fields.length == 0)
oid_field = fields.find { |field| field[:type] == OID_FIELD_TYPE }
if ids.length == 1
ids_field = { objectIds: ids.first }
else
if oid_field
# Note that ids is sorted
ids_field = { where: "#{oid_field[:name]} >=#{ids.first} AND #{oid_field[:name]} <=#{ids.last}" }
else
# This could be innefficient with large number of ids, but it is limited to MAX_BLOCK_SIZE
ids_field = { objectIds: ids.join(',') }
end
end
prepared_fields = fields.map { |field| "#{field[:name]}" }.join(',')
prepared_url = FEATURE_DATA_POST_URL % [url]
# @see http://resources.arcgis.com/en/help/arcgis-rest-api/index.html#/Query_Map_Service_Layer/02r3000000p1000000/
params_data = {
outFields: prepared_fields,
outSR: 4326,
f: 'json'
}
params_data.merge! ids_field
puts "#{prepared_url} (POST) Params:#{params_data}" if DEBUG
response = http_client.post(prepared_url, http_options(params_data, :post))
# Timeout connecting to ArcGIS
if response.code == 0
raise ExternalServiceError.new("TIMEOUT: #{prepared_url} : #{response.body} #{self.to_s}")
end
if response.code != 200
raise DataDownloadError.new("ERROR: #{prepared_url} POST " +
"#{params_data} (#{response.code}) : #{response.body} #{self.to_s}")
end
if response.code == 400 && !response.return_message.nil? \
&& response.return_message.downcase.include?('operation is not supported')
raise UnsupportedOperationError.new("#{request_url} (#{response.code}) : #{response.body}") \
end
begin
body = ::JSON.parse(response.body)
success = true
rescue JSON::ParserError
success = false
end
unless success
begin
# HACK: JSON spec does not cover Infinity
body = ::JSON.parse(response.body.gsub(':INF,', ':"Infinity",'))
rescue JSON::ParserError
raise ResponseError.new("JSON parsing error. URL: #{prepared_url} #{to_s}")
end
end
# Arcgis error
raise ExternalServiceError.new("#{prepared_url} : #{response.body}") if body.include?('error')
begin
retrieved_items = body.fetch('features')
return [] if retrieved_items.nil? || retrieved_items.empty?
retrieved_fields = body.fetch('fields')
geometry_type = body.fetch('geometryType')
spatial_reference = body.fetch('spatialReference')
rescue => exception
raise ResponseError.new("Missing data: #{exception.to_s} #{prepared_url} #{exception.backtrace}")
end
raise ResponseError.new("'fields' empty or invalid #{prepared_url}") \
if (retrieved_fields.nil? || retrieved_fields.length == 0)
raise ResponseError.new("'features' empty or invalid #{prepared_url}") \
if (retrieved_items.nil? || !retrieved_items.kind_of?(Array))
# Fields can be optional, cannot be enforced to always be present
desired_fields = fields.map { |field| field[:name] }
{
geometryType: geometry_type,
spatialReference: spatial_reference,
fields: retrieved_fields,
features: retrieved_items.collect { |item|
{
'attributes' => item['attributes'].select{ |k, v| desired_fields.include?(k) },
'geometry' => item['geometry']
}
}
}
end
# By default, will update the block size, incrementing or decrementing it according to stream operation results
# Block size only gets incremented after 2 successful streams to avoid scenario of:
# X items -> FAIL
# X/2 items -> PASS
# X items -> FAIL (again, because erroring item was at second half of X)
def block_size(update=true)
if update
if @current_stream_status && @last_stream_status && @block_size < MAX_BLOCK_SIZE
@block_size = [@block_size * BLOCK_FACTOR, MAX_BLOCK_SIZE].min
end
if !@current_stream_status && @block_size > MIN_BLOCK_SIZE
@block_size = [[(@block_size / BLOCK_FACTOR).floor, 1].max, MAX_BLOCK_SIZE].min
end
@block_size = [@block_size, @metadata[:max_records_per_query]].min
end
@block_size
end
def http_options(params={}, method=:get)
{
method: method,
params: method == :get ? params : {},
body: method == :post ? params : {},
followlocation: true,
ssl_verifypeer: false,
accept_encoding: 'gzip',
headers: { 'Accept-Charset' => 'utf-8' },
ssl_verifyhost: 0,
nosignal: true,
connecttimeout: HTTP_CONNECTION_TIMEOUT,
timeout: HTTP_TIMEOUT
}
end
def filename_from(feature_name)
feature_name.gsub(/[^\w]/, '_').downcase + '.json'
end
def validate_response(request_url, response)
raise ExternalServiceTimeoutError.new("TIMEOUT: #{request_url} : #{response.return_message}") \
if response.timed_out? || (response.code.zero? && !response.return_message.nil? \
&& response.return_message.downcase.include?('timeout'))
raise UnsupportedOperationError.new("#{request_url} (#{response.code}) : #{response.body}") \
if response.code == 400 && !response.return_message.nil? \
&& response.return_message.downcase.include?('operation is not supported')
raise DataDownloadError.new("#{request_url} (#{response.code}) : #{response.body}") \
if response.code != 200
end
end
end
class URLTooLargeError < StandardError
end
end
end
@@ -0,0 +1,543 @@
require_relative '../../../../../lib/carto/http/client'
module CartoDB
module Datasources
module Url
# BoxAPI module is a replacement for Boxr, which requires Ruby >= 2.
# Most of this code has been extracted from Boxr.
# This should be migrated when we upgrade to Ruby 2.
module BoxAPI
class ExpiredTokenError < StandardError; end
def self.oauth_url(state, options = {})
host = options.fetch(:host, "app.box.com")
response_type = options.fetch(:response_type, "code")
scope = options[:scope]
folder_id = options[:folder_id]
client_id = options[:client_id]
template = Addressable::Template.new("https://{host}/api/oauth2/authorize{?query*}")
query = { "response_type" => "#{response_type}", "state" => "#{state}", "client_id" => "#{client_id}" }
query["scope"] = "#{scope}" unless scope.nil?
query["folder_id"] = "#{folder_id}" unless folder_id.nil?
template.expand("host" => "#{host}", "query" => query)
end
def self.get_tokens(code, options = {})
grant_type = options[:grant_type]
assertion = options[:assertion]
scope = options[:scope]
username = options[:username]
client_id = options[:client_id]
client_secret = options[:client_secret]
uri = "https://api.box.com/oauth2/token"
body = "grant_type=#{grant_type}&client_id=#{client_id}&client_secret=#{client_secret}"
body = body + "&code=#{code}" unless code.nil?
body = body + "&scope=#{scope}" unless scope.nil?
body = body + "&username=#{username}" unless username.nil?
body = body + "&assertion=#{assertion}" unless assertion.nil?
auth_post(uri, body)
end
def self.refresh_tokens(refresh_token, options = {})
client_id = options[:client_id]
client_secret = options[:client_secret]
uri = "https://api.box.com/oauth2/token"
body = "grant_type=refresh_token&refresh_token=#{refresh_token}&client_id=#{client_id}&client_secret=#{client_secret}"
auth_post(uri, body)
end
def self.auth_post(uri, body, json_body = true)
uri = Addressable::URI.encode(uri)
res = post(uri, body: body)
if res.response_code == 200
json_body ? JSON.parse(res.response_body) : res.response_body
else
handle_error_response(res)
end
end
def self.handle_error_response(res)
body_json = JSON.parse(res.response_body)
if body_json['error'] == 'invalid_grant'
raise ExpiredTokenError.new(body_json.fetch('error_description', 'Expired token'))
else
raise "Box Error status: #{res.response_code}, body: #{res.response_body}, headers: #{res.response_headers}"
end
rescue => e
CartoDB.notify_exception(e, self: inspect, response: res.inspect)
raise e
end
def self.post(uri, options = {})
http_client = Carto::Http::Client.get('box',
connecttimeout: 60,
timeout: 600
)
http_client.post(uri.to_s, options)
end
def self.get(uri, options = {})
query = options[:query]
header = options[:header]
follow_redirect = options[:follow_redirect]
http_client = Carto::Http::Client.get('box',
connecttimeout: 60,
timeout: 600)
response = http_client.get(uri.to_s, headers: header, followlocation: follow_redirect, params: query)
response
end
end
module BoxAPI
class Client
API_URI = "https://api.box.com/2.0"
SEARCH_URI = "#{API_URI}/search"
FILES_URI = "#{API_URI}/files"
def initialize(access_token, options = {})
client_id = options[:client_id]
client_secret = options[:client_secret]
@access_token = access_token
raise "Access token cannot be nil" if @access_token.nil?
@client_id = client_id
@client_secret = client_secret
end
def search(query, options = {})
scope = options[:scope]
file_extensions = options[:file_extensions]
created_at_range = options[:created_at_range]
updated_at_range = options[:updated_at_range]
size_range = options[:size_range]
owner_user_ids = options[:owner_user_ids]
ancestor_folder_ids = options[:ancestor_folder_ids]
content_types = options[:content_types]
type = options[:type]
limit = options.fetch(:limit, 30)
offset = options.fetch(:offset, 0)
query = { query: query }
query[:scope] = scope unless scope.nil?
query[:file_extensions] = file_extensions unless file_extensions.nil?
query[:created_at_range] = created_at_range unless created_at_range.nil?
query[:updated_at_range] = updated_at_range unless updated_at_range.nil?
query[:size_range] = size_range unless size_range.nil?
query[:owner_user_ids] = owner_user_ids unless owner_user_ids.nil?
query[:ancestor_folder_ids] = ancestor_folder_ids unless ancestor_folder_ids.nil?
query[:content_types] = content_types unless content_types.nil?
query[:type] = type unless type.nil?
query[:limit] = limit unless limit.nil?
query[:offset] = offset unless offset.nil?
results, _response = get(SEARCH_URI, query: query)
results['entries']
end
def download_url(file, options = {})
version = options[:version]
download_file(file, version: version, follow_redirect: false)
end
def download_file(file, options = {})
version = options[:version]
follow_redirect = options.fetch(:follow_redirect, true)
file_id = ensure_id(file)
begin
uri = "#{FILES_URI}/#{file_id}/content"
query = {}
query[:version] = version unless version.nil?
# Boxr didn't have 200
_body_json, response = get(uri, query: query, success_codes: [302, 202, 200], follow_redirect: false, process_response: false)
if response.response_code == 302
location = response.header['Location'][0]
if follow_redirect
file, response = get(location, process_response: false)
else
return location # simply return the url
end
elsif response.response_code == 202
retry_after_seconds = response.header['Retry-After'][0]
sleep retry_after_seconds.to_i
elsif response.response_code == 200
file = response.response_body
end
end until file
file
end
FOLDER_AND_FILE_FIELDS = [:type, :id, :sequence_id, :etag, :name, :created_at, :modified_at, :description,
:size, :path_collection, :created_by, :modified_by, :trashed_at, :purged_at,
:content_created_at, :content_modified_at, :owned_by, :shared_link,
:folder_upload_email,
:parent, :item_status, :item_collection, :sync_state, :has_collaborations,
:permissions, :tags,
:sha1, :shared_link, :version_number, :comment_count, :lock, :extension,
:is_package,
:expiring_embed_link, :can_non_owners_invite]
FOLDER_AND_FILE_FIELDS_QUERY = FOLDER_AND_FILE_FIELDS.join(',')
def file_from_id(file_id, fields = [])
file_id = ensure_id(file_id)
uri = "#{FILES_URI}/#{file_id}"
query = build_fields_query(fields, FOLDER_AND_FILE_FIELDS_QUERY)
file, _response = get(uri, query: query)
file
end
# Required for all providers
DATASOURCE_NAME = 'box'
def revoke_tokens(token)
uri = "https://api.box.com/oauth2/revoke"
body = "client_id=#{@client_id}&client_secret=#{@client_secret}&token=#{token}"
BoxAPI::auth_post(uri, body, false)
rescue => ex
raise AuthError.new("revoke_token: #{ex.message}", DATASOURCE_NAME)
end
private
def build_fields_query(fields, all_fields_query)
if fields == :all
{ fields: all_fields_query }
elsif fields.is_a?(Array) && fields.length > 0
{ fields: fields.join(',') }
else
{}
end
end
def ensure_id(item)
return item if item.class == String || item.class == Integer || item.nil?
return item.id if item.respond_to?(:id)
return item['id'] if item.class == Hash
raise "Expecting an id of class String or Integer, or object that responds to :id"
end
def get(uri, options = {})
query = options[:query]
success_codes = options.fetch(:success_codes, [200])
process_response = options.fetch(:process_response, true)
if_match = options[:if_match]
box_api_header = options[:box_api_header]
follow_redirect = options.fetch(follow_redirect, true)
headers = standard_headers
headers['If-Match'] = if_match unless if_match.nil?
headers['BoxApi'] = box_api_header unless box_api_header.nil?
res = BoxAPI::get(uri, query: query, header: headers, follow_redirect: follow_redirect)
check_response_status(res, success_codes)
if process_response
return JSON.parse(res.response_body)
else
return res.response_body, res
end
end
def check_response_status(res, success_codes)
unless success_codes.include?(res.response_code)
raise "BoxError status: #{res.response_code}, body: #{res.response_body}, header: #{res.response_headers}"
end
end
def standard_headers
{ "Authorization" => "Bearer #{@access_token}" }
end
end
end
class Box < BaseOAuth
# Required for all providers
DATASOURCE_NAME = 'box'
# Constructor (hidden)
# @param config
# [
# 'application_name'
# 'client_id'
# 'client_secret'
# ]
# @param user ::User
# @throws UninitializedError
# @throws MissingConfigurationError
def initialize(config, user)
super(config, user, %w{ application_name client_id client_secret box_host }, DATASOURCE_NAME)
raise UninitializedError.new('missing user instance', DATASOURCE_NAME) if user.nil?
@access_token = nil
@refresh_token = nil
end
# Factory method
# @param config {}
# @param user ::User
# @return CartoDB::Datasources::Url::GDrive
def self.get_new(config, user)
new(config, user)
end
# If will provide a url to download the resource, or requires calling get_resource()
# @return bool
def providers_download_url?
false
end
# Return the url to be displayed or sent the user to to authenticate and get authorization code.
# Older implementations had a use_callback_flow parameter that became deprecated. Not implemented.
# @return string | nil
def get_auth_url
service_name = service_name_for_user(DATASOURCE_NAME, @user)
state = CALLBACK_STATE_DATA_PLACEHOLDER.sub('user', @user.username).sub('service', service_name)
BoxAPI::oauth_url(state,
host: config['box_host'],
response_type: "code",
scope: nil,
folder_id: nil,
client_id: config['client_id']).to_s
end
# Validates authorization code, sets access and refresh tokens for current instance and stores (refresh) token
# @param auth_code : string
# @return string : Refresh token
# Older implementations had a use_callback_flow parameter that became deprecated. Not implemented.
# @throws AuthError
def validate_auth_code(auth_code)
set_tokens(get_tokens(auth_code))
@refresh_token
end
# Validates the authorization callback
# @param params : mixed
def validate_callback(params)
if params[:error].present?
raise AuthError.new("validate_callback: #{params[:error]}", DATASOURCE_NAME)
end
if params[:code]
validate_auth_code(params[:code])
else
raise AuthError.new('validate_callback: Missing authorization code', DATASOURCE_NAME)
end
end
# Store (refresh) token. If it's not valid both access_token and refresh_token will be nil.
# Triggers generation of a valid access token for the lifetime of this instance
# @param token string
# @throws AuthError
def token=(token)
set_tokens(get_fresh_tokens(token))
rescue CartoDB::Datasources::Url::BoxAPI::ExpiredTokenError => e
CartoDB.notify_exception(e, self: inspect, token: token)
set_tokens('access_token' => nil, 'refresh_token' => nil)
end
# Retrieve (refresh) token
# @return string | nil
def token
@refresh_token
end
# Perform the listing and return results
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
# @return [ { :id, :title, :url, :service } ]
# @throws TokenExpiredOrInvalidError
# @throws DataDownloadError
def get_resources_list(filter = [])
self.filter = filter
# Box doesn't have a way to "retrieve everything" but it supports whitespaces for multiple search terms
result = client.search(supported_extensions.join(' '),
scope: nil,
file_extensions: nil,
created_at_range: nil,
updated_at_range: nil,
size_range: nil,
owner_user_ids: nil,
ancestor_folder_ids: nil,
content_types: nil,
type: nil,
limit: 200,
offset: 0)
result = result.map { |i| format_item_data(i) }.sort { |x, y| y[:updated_at] <=> x[:updated_at] }
unless @formats.nil? || @formats.empty?
result = result.select { |item| item[:filename] =~ /.*(#{@formats.join(')|(')})$/i }
end
result
end
# Retrieves a resource and returns its contents
# @param id string
# @return mixed
# @throws TokenExpiredOrInvalidError
# @throws DataDownloadError
def get_resource(id)
file = client.file_from_id(id)
client.download_file(file)
end
# @param id string
# @return Hash
# @throws TokenExpiredOrInvalidError
# @throws DataDownloadError
# @throws NotFoundDownloadError
def get_resource_metadata(id)
result = client.file_from_id(id)
if result.nil?
message = "Retrieving file #{id} metadata: #{result.inspect}, should stop syncing"
raise NotFoundDownloadError.new(message, DATASOURCE_NAME)
end
if result['item_status'] != 'active'
raise DataDownloadError.new("Retrieving file #{id} metadata: #{result.inspect}", DATASOURCE_NAME)
end
format_item_data(result)
end
# Retrieves current filters
# @return {}
def filter
@formats
end
# Sets current filters
# @param filter_data {}
def filter=(filter_data = [])
@formats = filter_data
end
# Just return datasource name
# @return string
def to_s
DATASOURCE_NAME
end
# If this datasource accepts a data import instance
# @return Boolean
def persists_state_via_data_import?
false
end
# Stores the data import item instance to use/manipulate it
# @param value DataImport
# Not implemented
def data_import_item=(_value)
nil
end
# Checks if token is still valid or has been revoked
# @return bool
# @throws AuthError
def token_valid?
raise 'invalid_token' unless token
# Any call would do, we just want to see if communicates or refuses the token
result = client.search('test search')
!result.nil?
rescue => e
if e.message =~ /invalid_token/
CartoDB.notify_debug('Box invalid_token', self: inspect)
false
else
CartoDB.notify_exception(e, self: inspect)
raise e
end
end
# Revokes current set token
def revoke_token
client.revoke_tokens(token)
end
private
def set_tokens(tokens)
@access_token = tokens['access_token']
@refresh_token = tokens['refresh_token']
end
def client
@client ||= get_client
end
def get_client
BoxAPI::Client.new(@access_token,
client_id: config['client_id'],
client_secret: config['client_secret'])
end
def get_tokens(code)
BoxAPI::get_tokens(code,
grant_type: "authorization_code",
assertion: nil,
scope: nil,
username: nil,
client_id: config['client_id'],
client_secret: config['client_secret'])
end
def get_fresh_tokens(refresh_token)
tokens = BoxAPI::refresh_tokens(refresh_token,
client_id: config['client_id'],
client_secret: config['client_secret'])
# Box refresh tokens can only be used once
update_user_oauth(tokens['refresh_token'])
tokens
end
def update_user_oauth(refresh_token)
carto_user = Carto::User.find(@user.id)
oauth = carto_user.oauth_for_service('box')
oauth.token = refresh_token
oauth.save
end
# Formats all data to comply with our desired format
# @param item_data Hash : Single item returned from GDrive API
# @return { :id, :title, :url, :service, :checksum, :size, :filename, :updated_at }
def format_item_data(item_data)
{
id: item_data['id'],
title: item_data['name'],
service: DATASOURCE_NAME,
checksum: checksum_of(item_data.fetch('modified_at')),
filename: item_data['name'],
size: item_data['size'].to_i,
updated_at: DateTime.rfc3339(item_data['content_modified_at'])
}
end
end
end
end
end
@@ -0,0 +1,281 @@
require 'dropbox_api'
require_relative '../base_oauth'
require_relative '../../../../../lib/dropbox_api/endpoints/auth/token/revoke'
module CartoDB
module Datasources
module Url
# In order to test Dropbox in local, do the following:
# 1. In Dropbox, change OAuth2 configuration, adding this (replace username and API key as needed):
# http://localhost:3000/u/juanignaciosl/api/v1/imports/service/dropbox/oauth_callback/?api_key=3312b39c6360862e13217a8aec540e57367f4a4b
# 2. Configure it in app_config.yml:
# dropbox:
# app_key: '528omteaww7fj86'
# app_secret: 'rhx2ovpuni266ra'
# callback_url: 'http://localhost:3000/u/juanignaciosl/api/v1/imports/service/dropbox/oauth_callback/?api_key=3312b39c6360862e13217a8aec540e57367f4a4b'
# This obviously will work for a single user.
class Dropbox < BaseOAuth
# Required for all datasources
DATASOURCE_NAME = 'dropbox'
# Specific of this datasource
FORMATS_TO_SEARCH_QUERIES = {
FORMAT_CSV => %W( .csv ),
FORMAT_EXCEL => %W( .xls .xlsx ),
FORMAT_GPX => %W( .gpx ),
FORMAT_KML => %W( .kml ),
FORMAT_COMPRESSED => %W( .zip )
}
# Constructor
# @param config Array
# [
# 'app_key'
# 'app_secret'
# 'callback_url'
# ]
# @param user ::User
# @throws UninitializedError
# @throws MissingConfigurationError
def initialize(config, user)
super(config, user, %w{ app_key app_secret callback_url }, DATASOURCE_NAME)
@user = user
@app_key = config.fetch('app_key')
@app_secret = config.fetch('app_secret')
@callback_url = config.fetch('callback_url')
self.filter = []
@access_token = nil
@auth_flow = nil
@client = nil
end
# Factory method
# @param config : {}
# @param user : ::User
# @return CartoDB::Datasources::Url::Dropbox
def self.get_new(config, user)
return new(config, user)
end
# If will provide a url to download the resource, or requires calling get_resource()
# @return bool
def providers_download_url?
false
end
# Return the url to be displayed or sent the user to to authenticate and get authorization code
# Older implementations had a use_callback_flow parameter that became deprecated. Not implemented.
# @throws AuthError
def get_auth_url
authenticator.authorize_url redirect_uri: @callback_url, state: state
rescue => ex
raise AuthError.new("get_auth_url(#{use_callback_flow}): #{ex.message}", DATASOURCE_NAME)
end
# Validates the authorization callback
# @param params : mixed
def validate_callback(params)
raise "state doesn't match" unless params[:state] == state
auth_bearer = authenticator.get_token(params[:code], redirect_uri: @callback_url)
@access_token = auth_bearer.token
@client = DropboxApi::Client.new(@access_token)
@access_token
rescue => ex
raise AuthError.new("validate_callback(#{params.inspect}): #{ex.message}", DATASOURCE_NAME)
end
# Set the token
# @param token string
# @throws TokenExpiredOrInvalidError
# @throws AuthError
def token=(token)
@access_token = token
@client = DropboxApi::Client.new(@access_token)
rescue => ex
handle_error(ex, "token= : #{ex.message}")
end
# Retrieve set token
# @return string | nil
def token
@access_token
end
# Perform the listing and return results
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
# @return [ { :id, :title, :url, :service } ]
# @throws TokenExpiredOrInvalidError
# @throws AuthError
# @throws DataDownloadError
def get_resources_list(filter=[])
all_results = []
self.filter = filter
@formats.each do |search_query|
start = 0
loop do
response = @client.search(search_query, '', max_results: SEARCH_BATCH_SIZE, start: start)
response.matches.select { |item| item.resource.is_a?(DropboxApi::Metadata::File) }.each do |item|
all_results.push(format_item_data(item.resource))
end
break unless response.has_more?
start += SEARCH_BATCH_SIZE
end
end
all_results
rescue => ex
handle_error(ex, "get_resources_list(): #{ex.message}")
end
# Retrieves a resource and returns its contents
# @param id string
# @return mixed
# @throws TokenExpiredOrInvalidError
# @throws AuthError
# @throws DataDownloadError
def get_resource(id)
file_contents = ''
@client.download(id) do |chunk|
file_contents << chunk
end
file_contents
rescue => ex
handle_error(ex, "get_resource() #{id}: #{ex.message}")
end
# @param id string
# @return Hash
# @throws TokenExpiredOrInvalidError
# @throws AuthError
# @throws DataDownloadError
def get_resource_metadata(id)
raise DropboxPermissionError.new('No Dropbox client', DATASOURCE_NAME) unless @client.present?
response = @client.get_metadata(id)
item_data = format_item_data(response)
item_data.to_hash
rescue => ex
handle_error(ex, "get_resource_metadata() #{id}: #{ex.message}")
end
# Retrieves current filters
# @return {}
def filter
@formats
end
# Sets current filters
# @param filter_data {}
def filter=(filter_data=[])
@formats = []
FORMATS_TO_SEARCH_QUERIES.each do |id, queries|
if filter_data.empty? || filter_data.include?(id)
queries.each do |query|
@formats = @formats.push(query)
end
end
end
end
# Just return datasource name
# @return string
def to_s
DATASOURCE_NAME
end
# If this datasource accepts a data import instance
# @return Boolean
def persists_state_via_data_import?
false
end
# Stores the data import item instance to use/manipulate it
# @param value DataImport
def data_import_item=(value)
nil
end
# Checks if token is still valid or has been revoked
# @return bool
# @throws AuthError
def token_valid?
# Any call would do, we just want to see if communicates or refuses the token
@client.get_current_account
true
rescue DropboxApi::Errors::HttpError => ex
CartoDB::Logger.debug(message: 'Invalid Dropbox token', exception: ex, user: @user)
false
end
# Revokes current set token
def revoke_token
@client.revoke
true
rescue DropboxApi::Errors::HttpError => ex
CartoDB::Logger.debug(message: 'Error revoking Dropbox token', exception: ex, user: @user)
true
rescue => ex
raise AuthError.new("revoke_token: #{ex.message}", DATASOURCE_NAME)
end
private
SEARCH_BATCH_SIZE = 1000
# Handles
# @param original_exception mixed
# @param message string
# @throws TokenExpiredOrInvalidError
# @throws AuthError
# @throws mixed
def handle_error(original_exception, message)
if original_exception.is_a? DropboxApi::Errors::NotFoundError
raise NotFoundDownloadError.new(message, DATASOURCE_NAME)
elsif original_exception.is_a? DropboxApi::Errors::BasicError
error_code = original_exception.http_response.code.to_i
if error_code == 401 || error_code == 403
raise TokenExpiredOrInvalidError.new(message, DATASOURCE_NAME)
else
raise AuthError.new(message)
end
elsif original_exception.is_a? ArgumentError
raise DataDownloadError.new(message, DATASOURCE_NAME)
else
raise original_exception
end
end
# Formats all data to comply with our desired format
# @param item_data Hash : Single item returned from Dropbox API
# @return { :id, :title, :url, :service, :size }
def format_item_data(resource)
path = resource.path_display
filename = path.split('/').last
{
id: path,
title: filename,
filename: filename,
service: DATASOURCE_NAME,
checksum: checksum_of(resource.rev),
size: resource.size
}
end
def authenticator
@authenticator ||= DropboxApi::Authenticator.new(@app_key, @app_secret)
end
def state
service_name = service_name_for_user(DATASOURCE_NAME, @user)
CALLBACK_STATE_DATA_PLACEHOLDER.sub('user', @user.username).sub('service', service_name)
end
end
end
end
end
@@ -0,0 +1,353 @@
require 'signet/oauth_2/client'
require 'google/apis/drive_v2'
require_relative '../../../../../lib/carto/http/client'
module CartoDB
module Datasources
module Url
class GDrive < BaseOAuth
# Required for all providers
DATASOURCE_NAME = 'gdrive'
OAUTH_SCOPES = ['https://www.googleapis.com/auth/drive'].freeze
# For when using authorization code instead of callback with token
REDIRECT_URI = 'urn:ietf:wg:oauth:2.0:oob'
FIELDS_TO_RETRIEVE = 'items(downloadUrl,exportLinks,id,modifiedDate,title,fileExtension,fileSize)'
# Specific of this provider
FORMATS_TO_MIME_TYPES = {
FORMAT_CSV => %w(text/csv),
FORMAT_EXCEL => %w(application/vnd.ms-excel application/vnd.google-apps.spreadsheet application/vnd.openxmlformats-officedocument.spreadsheetml.sheet),
# FORMAT_GPX => %w(text/xml), # Disabled because text/xml list any XML file
FORMAT_KML => %w(application/vnd.google-earth.kml+xml),
FORMAT_COMPRESSED => %w(application/zip application/x-zip-compressed), # application/x-compressed-tar application/x-gzip application/x-bzip application/x-tar )
}
# Constructor (hidden)
# @param config
# [
# 'application_name'
# 'client_id'
# 'client_secret'
# ]
# @param user ::User
# @throws UninitializedError
# @throws MissingConfigurationError
def initialize(config, user)
super(config, user, %w{ application_name client_id client_secret callback_url }, DATASOURCE_NAME)
raise UninitializedError.new('missing user instance', DATASOURCE_NAME) if user.nil?
self.filter=[]
@refresh_token = nil
@user = user
@callback_url = config.fetch('callback_url')
@client = Signet::OAuth2::Client.new(
authorization_uri: 'https://accounts.google.com/o/oauth2/auth',
token_credential_uri: 'https://oauth2.googleapis.com/token',
client_id: config.fetch('client_id'),
client_secret: config.fetch('client_secret'),
scope: OAUTH_SCOPES,
redirect_uri: @callback_url,
access_type: :offline
)
@drive = Google::Apis::DriveV2::DriveService.new
@drive.authorization = @client
end
# Factory method
# @param config {}
# @param user ::User
# @return CartoDB::Datasources::Url::GDrive
def self.get_new(config, user)
new(config, user)
end
# If will provide a url to download the resource, or requires calling get_resource()
# @return bool
def providers_download_url?
false
end
# Return the url to be displayed or sent the user to authenticate and get authorization code
# @param use_callback_flow : bool
# @return string | nil
def get_auth_url(use_callback_flow = true)
if use_callback_flow
service_name = service_name_for_user(DATASOURCE_NAME, @user)
@client.state = CALLBACK_STATE_DATA_PLACEHOLDER.sub('user', @user.username)
.sub('service', service_name)
else
@client.redirect_uri = REDIRECT_URI
end
@client.authorization_uri.to_s
end
# Validate authorization code and store token
# @param auth_code : string
# @param use_callback_flow : bool
# @return string : Access token
# @throws AuthError
def validate_auth_code(auth_code, use_callback_flow = true)
unless use_callback_flow
@client.redirect_uri = REDIRECT_URI
end
@client.code = auth_code
@client.fetch_access_token!
if @client.refresh_token.nil?
raise AuthError.new(
"Error validating auth token. Is this Google account linked to another CARTO account?",
DATASOURCE_NAME
)
end
@refresh_token = @client.refresh_token
rescue Google::Apis::AuthorizationError, Signet::AuthorizationError => ex
raise AuthError.new("validating auth code: #{ex.message}", DATASOURCE_NAME)
end
# Validates the authorization callback
# @param params : mixed
def validate_callback(params)
if params[:error].present?
raise AuthError.new("validate_callback: #{params[:error]}", DATASOURCE_NAME)
end
if params[:code]
validate_auth_code(params[:code])
else
raise AuthError.new('validate_callback: Missing authorization code', DATASOURCE_NAME)
end
end
# Store token
# Triggers generation of a valid access token for the lifetime of this instance
# @param token string
# @throws AuthError
def token=(token)
@refresh_token = token
@client.update_token!(refresh_token: @refresh_token)
@client.fetch_access_token!
rescue Google::Apis::AuthorizationError, Signet::AuthorizationError => ex
raise TokenExpiredOrInvalidError.new("Invalid token: #{ex.message}", DATASOURCE_NAME)
rescue Google::Apis::ClientError, \
Google::Apis::ServerError, Google::Apis::BatchError, Google::Apis::TransmissionError => ex
raise AuthError.new("setting token: #{ex.message}", DATASOURCE_NAME)
end
# Retrieve token
# @return string | nil
def token
@refresh_token
end
# Perform the listing and return results
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
# @return [ { :id, :title, :url, :service } ]
# @throws TokenExpiredOrInvalidError
# @throws DataDownloadError
def get_resources_list(filter=[])
all_results = []
self.filter = filter
@drive.batch do |d|
@formats.each do |mime_type|
d.list_files(q: "mime_type = '#{mime_type}'", fields: FIELDS_TO_RETRIEVE) do |res, err|
if err
case err.status_code
when 200
break
when 403
raise GDriveNoExternalAppsAllowedError.new(result.data['error']['message'], DATASOURCE_NAME)
else
error_msg = "get_resources_list() #{result.data['error']['message']} (#{result.status})"
raise DataDownloadError.new(error_msg, DATASOURCE_NAME)
end
elsif res.items.present?
res.items.each do |item|
all_results.push(format_item_data(item))
end
end
end
end
end
all_results.compact
rescue Google::Apis::AuthorizationError, Signet::AuthorizationError => ex
raise TokenExpiredOrInvalidError.new("Invalid token: #{ex.message}", DATASOURCE_NAME)
rescue Google::Apis::BatchError, Google::Apis::TransmissionError, Google::Apis::ClientError, \
Google::Apis::ServerError => ex
raise DataDownloadError.new("getting resources: #{ex.message}", DATASOURCE_NAME)
end
# Retrieves a resource and returns its contents
# @param id string
# @return mixed
# @throws TokenExpiredOrInvalidError
# @throws DataDownloadError
def get_resource(id)
@drive.get_file(id) do |file, err|
if err
error_msg = "(#{err.status_code}) retrieving file #{id}: #{err}"
raise DataDownloadError.new(error_msg, DATASOURCE_NAME)
end
if file.export_links.present?
@drive.export_file(file.id, 'text/csv', download_dest: StringIO.new) do |content, export_err|
raise export_err if export_err
return content
end
else
@drive.get_file(file.id, download_dest: StringIO.new) do |content, download_err|
raise download_err if download_err
return content
end
end
end
rescue Google::Apis::AuthorizationError, Signet::AuthorizationError => ex
raise TokenExpiredOrInvalidError.new("Invalid token: #{ex.message}", DATASOURCE_NAME)
rescue Google::Apis::BatchError, Google::Apis::TransmissionError, Google::Apis::ClientError, \
Google::Apis::ServerError => ex
raise DataDownloadError.new("downloading file #{id}: #{ex.message}", DATASOURCE_NAME)
end
# @param id string
# @return Hash
# @throws TokenExpiredOrInvalidError
# @throws DataDownloadError
# @throws NotFoundDownloadError
def get_resource_metadata(id)
@drive.get_file(id) do |file, err|
if err
case err.status_code
when 404
error_msg = "(#{err.status_code}) retrieving file #{id} metadata: #{err}, should stop syncing"
raise NotFoundDownloadError.new(error_msg, DATASOURCE_NAME)
else
error_msg = "(#{err.status_code}) retrieving file #{id} metadata: #{err}"
raise DataDownloadError.new(error_msg, DATASOURCE_NAME)
end
else
item_data = format_item_data(file)
return item_data.to_hash
end
end
rescue Google::Apis::AuthorizationError, Signet::AuthorizationError
raise TokenExpiredOrInvalidError.new('Invalid token', DATASOURCE_NAME)
rescue Google::Apis::BatchError, Google::Apis::TransmissionError, Google::Apis::ClientError, \
Google::Apis::ServerError
raise DataDownloadError.new("get_resource_metadata() #{id}", DATASOURCE_NAME)
rescue StandardError => e
CartoDB.notify_exception(e, id: id, user: @user)
raise e
end
# Retrieves current filters
# @return {}
def filter
@formats
end
# Sets current filters
# @param filter_data {}
def filter=(filter_data=[])
@formats = []
FORMATS_TO_MIME_TYPES.each do |id, mime_types|
if filter_data.empty? || filter_data.include?(id)
mime_types.each do |mime_type|
@formats = @formats.push(mime_type)
end
end
end
end
# Just return datasource name
# @return string
def to_s
DATASOURCE_NAME
end
# If this datasource accepts a data import instance
# @return Boolean
def persists_state_via_data_import?
false
end
# Stores the data import item instance to use/manipulate it
# @param value DataImport
def data_import_item=(value)
nil
end
# Checks if token is still valid or has been revoked
# @return bool
# @throws AuthError
def token_valid?
# Any call would do, we just want to see if communicates or refuses the token
result = @drive.get_about
!result.nil?
rescue Google::Apis::AuthorizationError, Signet::AuthorizationError
false
rescue Google::Apis::BatchError, Google::Apis::TransmissionError, Google::Apis::ClientError, \
Google::Apis::ServerError => ex
raise AuthError.new("token_valid?() #{id}: #{ex.message}", DATASOURCE_NAME)
end
# Revokes current set token
def revoke_token
http_client = Carto::Http::Client.get('gdrive',
connecttimeout: 60,
timeout: 600)
response = http_client.get("https://accounts.google.com/o/oauth2/revoke?token=#{token}")
if response.code == 200
true
end
rescue StandardError => ex
raise AuthError.new("revoke_token: #{ex.message}", DATASOURCE_NAME)
end
private
# Formats all data to comply with our desired format
# @param item_data Hash : Single item returned from GDrive API
# @return { :id, :title, :url, :service, :checksum, :size }
def format_item_data(item_data)
data =
{
id: item_data.id,
title: item_data.title,
service: DATASOURCE_NAME,
checksum: checksum_of(item_data.modified_date.to_s)
}
if item_data.export_links.present?
# Native spreadsheets have no format nor direct download links
data[:url] = item_data.export_links.first.last
data[:url] = data[:url][0..data[:url].rindex('=')] + 'csv'
data[:filename] = clean_filename(item_data.title) + '.csv'
data[:size] = NO_CONTENT_SIZE_PROVIDED
elsif item_data.download_url.present?
data[:url] = item_data.download_url
# For Drive files, title == filename + extension
data[:filename] = item_data.title
data[:size] = item_data.file_size.to_i
else
# Downloads from files shared by other people can be disabled, ignore them
return nil
end
data
end
def clean_filename(name)
clean_name = ''
name.gsub(' ','_').scan(/([a-zA-Z0-9_]+)/).flatten.map { |match|
clean_name << match
}
clean_name = name if clean_name.size == 0
clean_name
end
end
end
end
end
@@ -0,0 +1,287 @@
require "instagram"
module CartoDB
module Datasources
module Url
class InstagramOAuth < BaseOAuth
# Required for all datasources
DATASOURCE_NAME = 'instagram'
FORMAT_ALL_MEDIA = 'all_media'
# Constructor
# @param config Array
# [
# 'app_key'
# 'app_secret'
# 'callback_url'
# ]
# @param user ::User
# @throws UninitializedError
# @throws MissingConfigurationError
def initialize(config, user)
super(config, user, %w{ app_key app_secret callback_url }, DATASOURCE_NAME)
@user = user
@app_key = config.fetch('app_key')
@app_secret = config.fetch('app_secret')
raise ServiceDisabledError.new(DATASOURCE_NAME, @user.username) unless @user.has_feature_flag?('instagram_import')
service_name = service_name_for_user(DATASOURCE_NAME, @user)
placeholder = CALLBACK_STATE_DATA_PLACEHOLDER.sub('user', @user.username).sub('service', service_name)
@callback_url = "#{config.fetch('callback_url')}?state=#{placeholder}"
self.filter = []
@access_token = nil
@auth_flow = nil
@client = nil
end
# Factory method
# @param config : {}
# @param user : ::User
# @return CartoDB::Datasources::Url::InstagramOAuth
def self.get_new(config, user)
return new(config, user)
end
# If will provide a url to download the resource, or requires calling get_resource()
# @return bool
def providers_download_url?
false
end
# Return the url to be displayed or sent the user to to authenticate and get authorization code
# @param use_callback_flow : bool
# @throws AuthError
def get_auth_url(use_callback_flow=true)
# TODO: Add CSRF here (http://instagram.com/developer/authentication/)
Instagram.authorize_url({
client_id: @app_key,
response_type: 'code',
redirect_uri: @callback_url
})
rescue => ex
raise AuthError.new("get_auth_url(#{use_callback_flow}): #{ex.message}", DATASOURCE_NAME)
end
# Validates the authorization callback
# @param params : mixed
def validate_callback(params)
response = Instagram.get_access_token(params[:code], {
client_id: @app_key,
client_secret: @app_secret,
redirect_uri: @callback_url
})
@access_token = response.access_token
@client = Instagram.client(access_token: @access_token)
@access_token
rescue => ex
raise AuthError.new("validate_callback(#{params.inspect}): #{ex.message}", DATASOURCE_NAME)
end
# Set the token
# @param token string
# @throws TokenExpiredOrInvalidError
# @throws AuthError
def token=(token)
@access_token = token
@client = Instagram.client(access_token: @access_token)
rescue => ex
handle_error(ex, "token= : #{ex.message}")
end
# Retrieve set token
# @return string | nil
def token
@access_token
end
# Perform the listing and return results
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
# @return [ { :id, :title, :url, :service } ]
# @throws TokenExpiredOrInvalidError
# @throws AuthError
# @throws DataDownloadError
def get_resources_list(filter=[])
[
{
id: FORMAT_ALL_MEDIA,
title: 'All your photos and videos',
url: 'All your photos and videos',
service: DATASOURCE_NAME,
checksum: '',
size: NO_CONTENT_SIZE_PROVIDED
}
]
end
# Retrieves a resource and returns its contents
# @param id string
# @return mixed
# @throws TokenExpiredOrInvalidError
# @throws AuthError
# @throws DataDownloadError
def get_resource(id)
contents = [
field_to_csv('thumbnail'),
field_to_csv('image'),
field_to_csv('link'),
field_to_csv('type'),
field_to_csv('lat'),
field_to_csv('lon'),
field_to_csv('location_id'),
field_to_csv('location_name'),
field_to_csv('caption'),
field_to_csv('comments_count'),
field_to_csv('likes_count'),
field_to_csv('tags'),
field_to_csv('created_time')
].join(',') << "\n"
max_id = nil
begin
batch_contents, max_id = get_resource_page(id, max_id)
contents << batch_contents
end while !max_id.nil?
contents
end
# @param id string
# @return Hash
# @throws TokenExpiredOrInvalidError
# @throws AuthError
# @throws DataDownloadError
def get_resource_metadata(id)
{
id: FORMAT_ALL_MEDIA,
filename: "#{DATASOURCE_NAME}_#{@client.user.username}.csv",
size: NO_CONTENT_SIZE_PROVIDED
}
rescue => ex
handle_error(ex, "get_resource_metadata() #{id}: #{ex.message}")
end
# Retrieves current filters
# @return {}
def filter
{}
end
# Sets current filters
# @param filter_data {}
def filter=(filter_data=[])
nil
end
# Just return datasource name
# @return string
def to_s
DATASOURCE_NAME
end
# If this datasource accepts a data import instance
# @return Boolean
def persists_state_via_data_import?
false
end
# Stores the data import item instance to use/manipulate it
# @param value DataImport
def data_import_item=(value)
nil
end
# Checks if token is still valid or has been revoked
# @return bool
# @throws AuthError
def token_valid?
# checking if metadata is returned, if so token
# is valid, if not it is invalid
response = get_resource_metadata(DATASOURCE_NAME)
if response[:id]
true
end
rescue => ex
false
end
# Revokes current set token
def revoke_token
# TODO: See how to check this
true
rescue => ex
raise AuthError.new("revoke_token: #{ex.message}", DATASOURCE_NAME)
end
private
def field_to_csv(field)
'"' + field.to_s.gsub('"', '""').gsub("\\n", ' ').gsub("\x0D", ' ').gsub("\x0A", ' ').gsub("\0", '')
.gsub("\\", ' ') + '"'
end
# @param resource_id String
# @para max_id Integer|nil Max media id retrieved (used to paginate)
def get_resource_page(resource_id, max_id=nil)
contents = ''
data = { count: 30 }
data[:max_id] = max_id unless max_id.nil?
items = @client.user_recent_media(data)
new_max_id = items.pagination.next_max_id
items.each do |item|
if item.location.nil?
lat = lon = location_id = location_name = nil
else
lat = item.location.latitude
lon = item.location.longitude
location_id = item.location.id
location_name = item.location.name
end
caption = item.caption.nil? ? '' : item.caption.text
contents << [
field_to_csv(item.images.thumbnail.url),
field_to_csv(item.images.standard_resolution.url),
field_to_csv(item.link),
field_to_csv(item.type),
field_to_csv(lat),
field_to_csv(lon),
field_to_csv(location_id),
field_to_csv(location_name),
field_to_csv(caption),
field_to_csv(item.comments['count']),
field_to_csv(item.likes['count']),
field_to_csv(item.tags.join(',')),
field_to_csv(item.created_time)
].join(',') << "\n"
end
[ contents, new_max_id ]
rescue => ex
handle_error(ex, "get_resource() #{resource_id}: #{ex.message}")
end
# Handles
# @param original_exception mixed
# @param message string
# @throws TokenExpiredOrInvalidError
# @throws AuthError
# @throws mixed
def handle_error(original_exception, message)
# TODO: Implement
raise original_exception
end
end
end
end
end
@@ -0,0 +1,397 @@
require 'json'
require 'gibbon'
require 'addressable/uri'
require_relative '../base_oauth'
require_relative '../../../../../lib/carto/http/client'
module CartoDB
module Datasources
module Url
# Note:
# - MailChimp access tokens don't expire, no need to handle that logic
class MailChimp < BaseOAuth
# Required for all datasources
DATASOURCE_NAME = 'mailchimp'
AUTHORIZE_URI = 'https://login.mailchimp.com/oauth2/authorize?response_type=code&client_id=%s&redirect_uri=%s'
ACCESS_TOKEN_URI = 'https://login.mailchimp.com/oauth2/token'
MAILCHIMP_METADATA_URI = 'https://login.mailchimp.com/oauth2/metadata'
API_TIMEOUT_SECS = 60
# Constructor
# @param config Array
# [
# 'api_key'
# 'timeout_minutes'
# ]
# @param user ::User
# @throws UninitializedError
# @throws MissingConfigurationError
def initialize(config, user)
super(config, user, %w{ app_key app_secret callback_url }, DATASOURCE_NAME)
@user = user
@app_key = config.fetch('app_key')
@app_secret = config.fetch('app_secret')
@http_timeout = config.fetch(:http_timeout, 600)
@http_connect_timeout = config.fetch(:http_connect_timeout, 60)
service_name = service_name_for_user(DATASOURCE_NAME, @user)
placeholder = CALLBACK_STATE_DATA_PLACEHOLDER.sub('user', @user.username).sub('service', service_name)
@callback_url = "#{config.fetch('callback_url')}?state=#{placeholder}"
Gibbon::API.timeout = API_TIMEOUT_SECS
Gibbon::API.throws_exceptions = true
Gibbon::Export.timeout = API_TIMEOUT_SECS
Gibbon::Export.throws_exceptions = false
@access_token = nil
@api_client = nil
end
# Factory method
# @param config : {}
# @param user : ::User
# @return CartoDB::Datasources::Url::MailChimpLists
def self.get_new(config, user)
return new(config, user)
end
# If will provide a url to download the resource, or requires calling get_resource()
# @return bool
def providers_download_url?
false
end
# Return the url to be displayed or sent the user to to authenticate and get authorization code
# @param use_callback_flow : bool
# @return string : URL to navigate to for the authorization flow
# @throws ExternalServiceError
def get_auth_url(use_callback_flow=true)
if use_callback_flow
AUTHORIZE_URI % [@app_key, Addressable::URI.encode(@callback_url)]
else
raise ExternalServiceError.new("This datasource doesn't allows non-callback flows", DATASOURCE_NAME)
end
end
# Validate authorization code and store token
# @param auth_code : string
# @return string : Access token
# @throws ExternalServiceError
def validate_auth_code(auth_code)
raise ExternalServiceError.new("This datasource doesn't allows non-callback flows", DATASOURCE_NAME)
end
# Validates the authorization callback
# @param params : mixed
# @throws AuthError
# @throws DataDownloadTimeoutError
def validate_callback(params)
code = params.fetch('code')
if code.nil? || code == ''
raise "Empty callback code"
end
token_call_params = {
grant_type: 'authorization_code',
client_id: @app_key,
client_secret: @app_secret,
code: code,
redirect_uri: @callback_url
}
token_response = http_client.post(ACCESS_TOKEN_URI, http_options(token_call_params, :post))
raise DataDownloadTimeoutError.new(DATASOURCE_NAME) if token_response.timed_out?
unless token_response.code == 200
raise "Bad token response: #{token_response.body.inspect} (#{token_response.code})"
end
token_data = ::JSON.parse(token_response.body)
partial_access_token = token_data['access_token']
# Afterwards, must do another call to metadata endpoint to retrieve API details
# @see https://apidocs.mailchimp.com/oauth2/
metadata_response = http_client.get(MAILCHIMP_METADATA_URI,http_options({}, :get, {
'Authorization' => "OAuth #{partial_access_token}"}))
raise DataDownloadTimeoutError.new(DATASOURCE_NAME) if metadata_response.timed_out?
unless metadata_response.code == 200
raise "Bad metadata response: #{metadata_response.body.inspect} (#{metadata_response.code})"
end
metadata_data = ::JSON.parse(metadata_response.body)
# This specially formed token behaves as an API Key for client calls using API
@access_token = "#{partial_access_token}-#{metadata_data['dc']}"
rescue => ex
raise AuthError.new("validate_callback(#{params.inspect}): #{ex.message}", DATASOURCE_NAME)
end
# Set the token
# @param token string
# @throws TokenExpiredOrInvalidError
def token=(token)
@access_token = token
@api_client = Gibbon::API.new(@access_token)
rescue Gibbon::MailChimpError => exception
raise TokenExpiredOrInvalidError.new("token=() : #{exception.message} (API code: #{exception.code})",
DATASOURCE_NAME)
rescue => exception
raise TokenExpiredOrInvalidError.new("token=() : #{exception.inspect}", DATASOURCE_NAME)
end
# Retrieve set token
# @return string | nil
def token
@access_token
end
# Perform the listing and return results
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
# @return [ { :id, :title, :url, :service } ]
# @throws UninitializedError
# @throws DataDownloadError
def get_resources_list(filter=[])
raise UninitializedError.new('No API client instantiated', DATASOURCE_NAME) unless @api_client.present?
all_results = []
offset = 0
limit = 100
total = nil
begin
response = @api_client.campaigns.list({
start: offset,
limit: limit
})
errors = response.fetch('errors', [])
unless errors.empty?
raise DataDownloadError.new("get_resources_list(): #{errors.inspect}", DATASOURCE_NAME)
end
total = response.fetch('total', 0).to_i if total.nil?
response_data = response.fetch('data', [])
response_data.each do |item|
# Skip items without tracking
all_results.push(format_activity_item_data(item)) if item['tracking']['opens']
end
offset += limit
end while offset < total
all_results
rescue Gibbon::MailChimpError => exception
raise DataDownloadError.new("get_resources_list(): #{exception.message} (API code: #{exception.code}",
DATASOURCE_NAME)
rescue => exception
raise DataDownloadError.new("get_resources_list(): #{exception.inspect}", DATASOURCE_NAME)
end
# Retrieves a resource and returns its contents
# @param id string
# @return mixed
# @throws UninitializedError
# @throws DataDownloadError
def get_resource(id)
raise UninitializedError.new('No API client instantiated', DATASOURCE_NAME) unless @api_client.present?
subscribers = {}
contents = StringIO.new
export_api = @api_client.get_exporter
# 1) Retrieve campaign details
campaign = get_resource_metadata(id)
campaign_details = export_api.list({id: campaign[:list_id]})
campaign = nil
# 2) Retrieve subscriber activity
# https://apidocs.mailchimp.com/export/1.0/campaignsubscriberactivity.func.php
subscribers_activity = export_api.campaign_subscriber_activity({id: id})
subscribers_activity.each { |line|
store_subscriber_if_opened(line, subscribers)
}
subscribers_activity = nil
# 3) Update campaign details with subscriber activity results
# 4) anonymize data (inside list_json_to_csv)
campaign_details.each_with_index { |line, index|
contents.write list_json_to_csv(line, subscribers, index == 0)
}
contents.string
rescue Gibbon::MailChimpError => exception
raise DataDownloadError.new("get_resource(): #{exception.message} (API code: #{exception.code}",
DATASOURCE_NAME)
rescue => exception
raise DataDownloadError.new("get_resource(): #{exception.inspect}", DATASOURCE_NAME)
end
# @param id string
# @return Hash
# @throws UninitializedError
# @throws DataDownloadError
def get_resource_metadata(id)
raise UninitializedError.new('No API client instantiated', DATASOURCE_NAME) unless @api_client.present?
item_data = {}
# No metadata call at API, so just retrieve same info but from specific campaign id
# https://apidocs.mailchimp.com/api/2.0/campaigns/list.php
response = @api_client.campaigns.list({ filters: { campaign_id: id } })
errors = response.fetch('errors', [])
unless errors.empty?
raise DataDownloadError.new("get_resources_list(): #{errors.inspect}", DATASOURCE_NAME)
end
response_data = response.fetch('data', [])
response_data.each do |item|
if item.fetch('id') == id
item_data = format_activity_item_data(item)
end
end
item_data
rescue Gibbon::MailChimpError => exception
raise DataDownloadError.new("get_resource_metadata(): #{exception.message} (API code: #{exception.code}",
DATASOURCE_NAME)
rescue => exception
raise DataDownloadError.new("get_resource_metadata(): #{exception.inspect}", DATASOURCE_NAME)
end
# Retrieves current filters
# @return {}
def filter
[]
end
# Sets current filters
# @param filter_data {}
def filter=(filter_data=[])
end
# Just return datasource name
# @return string
def to_s
DATASOURCE_NAME
end
# If this datasource accepts a data import instance
# @return Boolean
def persists_state_via_data_import?
false
end
# Stores the data import item instance to use/manipulate it
# @param value DataImport
def data_import_item=(value)
nil
end
# Checks if token is still valid or has been revoked
# @return bool
# @throws AuthError
def token_valid?
raise UninitializedError.new('No API client instantiated', DATASOURCE_NAME) unless @api_client.present?
# Any call would do, we just want to see if communicates or refuses the token
# This call is available to all roles
response = @api_client.users.profile
# 'errors' only appears in failure scenarios, while 'username' only if went ok
response.fetch('errors', nil).nil? && !response.fetch('username', nil).nil?
rescue => ex
CartoDB.notify_exception(ex)
false
end
# Revokes current set token
def revoke_token
# not supported
end
private
def http_client
@http_client ||= Carto::Http::Client.get('mailchimp')
end
def http_options(params={}, method=:get, extra_headers={})
{
method: method,
params: method == :get ? params : {},
body: method == :post ? params : {},
followlocation: true,
ssl_verifypeer: false,
headers: {
'Accept' => 'application/json'
}.merge(extra_headers),
ssl_verifyhost: 0,
connecttimeout: @http_connect_timeout,
timeout: @http_timeout
}
end
# Formats all data to comply with our desired format
# @param item_data Hash : Single item returned from MailChimp API
# @return { :id, :title, :url, :service, :size }
def format_activity_item_data(item_data)
filename = item_data.fetch('title').gsub(' ', '_')
{
id: item_data.fetch('id'),
list_id: item_data.fetch('list_id'),
title: "#{item_data.fetch('title')}",
filename: "#{filename}.csv",
service: DATASOURCE_NAME,
checksum: '',
member_count: item_data.fetch('emails_sent'),
size: NO_CONTENT_SIZE_PROVIDED
}
end
def store_subscriber_if_opened(input_fields='[]', subscribers)
contents = ::JSON.parse(input_fields)
contents.each { |subject, actions|
unless actions.length == 0
actions.each { |action|
if action["action"] == "open"
subscribers[subject] = true
end
opened_action = true
}
end
}
end
# @param contents String containing a JSON array of fields (data of campaign user/target)
# @param subscribers Hash { subject => opened_email }
# @param header_row Boolean
# @return String Containing a CSV ready to dump to a file
def list_json_to_csv(contents='[]', subscribers={}, header_row=false)
# shorcut: Remove newlines and Anonymize email addresses before parsing to speed up
contents = ::JSON.parse(contents.gsub("\n", ' ').gsub(/(\w|\.|\-)+@/, ""))
opened_mail = !subscribers[contents[0]].nil?
cleaned_contents = []
#Once parsed, each row contains data like account code, company name, email, first name...
contents.each_with_index { |field, index|
# Remove double quotes to avoid CSV errors
cleaned_contents[index] = "\"#{field.to_s.gsub('"', '""')}\""
}
cleaned_contents.push("\"#{header_row ? 'Opened' : opened_mail.to_s}\"")
data = cleaned_contents.join(',')
data << "\n"
end
end
end
end
end
@@ -0,0 +1,181 @@
require_relative '../../../../../lib/carto/http/client'
module CartoDB
module Datasources
module Url
class PublicUrl < Base
# Required for all datasources
DATASOURCE_NAME = 'public_url'
URL_REGEXP = %r{://}
# Constructor (hidden)
# @param config
# [ ]
def initialize(config)
super
@http_timeout = config.fetch(:http_timeout, 3200)
@http_connect_timeout = config.fetch(:http_connect_timeout, 60)
@service_name = DATASOURCE_NAME
@headers = nil
@response = nil
end
# Factory method
# @param config {}
# @return CartoDB::Datasources::Url::PublicUrl
def self.get_new(config={})
return new(config)
end
# If will provide a url to download the resource, or requires calling get_resource()
# @return bool
def providers_download_url?
true
end
def get_http_response_code
@response.code if !@response.nil? && !@response.code.nil?
end
# Perform the listing and return results
# @param filter Array : (Optional) filter to specify which resources to retrieve. Leave empty for all supported.
# @return [ { :id, :title, :url, :service } ]
def get_resources_list(filter=[])
nil
end
# Retrieves a resource and returns its contents
# @param id string
# @return mixed
# @throws DataDownloadTimeoutError
# @throws DataDownloadError
def get_resource(id)
response = http_client.get(id, http_options)
while response.headers['location']
response = http_client.get(id, http_options)
end
raise DataDownloadTimeoutError.new(DATASOURCE_NAME) if response.timed_out?
raise DataDownloadError.new("get_resource() #{id}", DATASOURCE_NAME) unless response.code.to_s =~ /\A[23]\d+/
# To be used in when try to retrieve the http response code
@response = response
response.response_body
end
# @param id string
# @return Hash
def get_resource_metadata(id)
fetch_headers(id)
{
id: id,
title: id,
url: id,
service: DATASOURCE_NAME,
checksum: checksum_of(id, etag_header, last_modified_header),
size: content_length_header
# No need to use :filename nor file
}
end
# Fetches the headers for a given url
# @throws DataDownloadError
def fetch_headers(url)
if url =~ URL_REGEXP
response = http_client.head(url, http_options)
raise DataDownloadTimeoutError.new(DATASOURCE_NAME) if response.timed_out?
# For example S3 only allows one verb per signed url (we use GET) so won't allow HEAD, but it's ok
@headers = (response.code.to_s =~ /\A[23]\d+/) ? response.headers : {}
else
@headers = {}
end
end
# Get the etag header if present
# @return string
# @throws UninitializedError
def etag_header
raise UninitializedError.new('headers not fetched', DATASOURCE_NAME) if @headers.nil?
etag = @headers.fetch('ETag', nil)
etag ||= @headers.fetch('Etag', nil)
etag ||= @headers.fetch('etag', '')
etag = etag.delete('"').delete("'") unless etag.empty?
etag
end
# Get the last modified header if present
# @return string
# @throws UninitializedError
def last_modified_header
raise UninitializedError.new('headers not fetched', DATASOURCE_NAME) if @headers.nil?
last_modified = @headers.fetch('Last-Modified', nil)
last_modified ||= @headers.fetch('Last-modified', nil)
last_modified ||= @headers.fetch('last-modified', '')
last_modified = last_modified.delete('"').delete("'") unless last_modified.empty?
last_modified
end
# Just return datasource name
# @return string
def to_s
DATASOURCE_NAME
end
# If this datasource accepts a data import instance
# @return Boolean
def persists_state_via_data_import?
false
end
# Stores the data import item instance to use/manipulate it
# @param value DataImport
def data_import_item=(value)
nil
end
private
def http_client
@http_client ||= Carto::Http::Client.get('public_url')
end
# Get the file size if present
# @return Integer
# @throws UninitializedError
def content_length_header
raise UninitializedError.new('headers not fetched', DATASOURCE_NAME) if @headers.nil?
content_length = @headers.fetch('Content-Length', nil)
content_length ||= @headers.fetch('Content-length', nil)
content_length ||= @headers.fetch('content-length', NO_CONTENT_SIZE_PROVIDED)
content_length.to_i
end
# Calculates a checksum of given url
# @return string
def checksum_of(url, etag, last_modified)
#noinspection RubyArgCount
Zlib::crc32(url + etag + last_modified).to_s
end
# HTTP (Typhoeus) options
def http_options
{
followlocation: true,
ssl_verifypeer: false,
ssl_verifyhost: 0,
timeout: @http_timeout,
connecttimeout: @http_connect_timeout
}
end
end
end
end
end
@@ -0,0 +1,169 @@
require 'tempfile'
require 'fileutils'
require 'json'
module CartoDB
module Datasources
# TODO: Error handling, now assumes all done ok
class CSVFileDumper
ORIGINAL_FILE_EXTENSION = '.json'
CONVERTED_FILE_EXTENSION = '.csv'
HEADERS_FILE_EXTENSION = '_headers.csv'
FILE_DUMPER_TMP_SUBFOLDER = '/tmp/csv_file_dumper/'
OUTPUT_ENCODING = 'utf-8'
def initialize(json_to_csv_conversor, debug_mode = false)
@debug_mode = debug_mode
@json2csv_conversor = json_to_csv_conversor
@temporary_directory = nil
@temporary_folder = Time.now.strftime("%Y%m%d_%H%M%S_") + rand(1000).to_s
@additional_fields = {}
@files = {}
@original_files = {}
@headers_file = nil
@buffer_size = 8192
end
def buffer_size=(value)
@buffer_size = value.to_i if value.to_i > 0
end
def additional_fields=(data = {})
@additional_fields = data
end
# This class uses a temporal CSV file per name
# optionally dumping also the source JSON in another file if in debug mode
# @param name String
def begin_dump(name)
# Create temp file & open
@files[name] = temporary_file(name)
@original_files[name] = temporary_file(name, ORIGINAL_FILE_EXTENSION) if @debug_mode
@headers_file = temporary_file('', HEADERS_FILE_EXTENSION) if @debug_mode
end
# @param name String
# @param data Array
# @return Integer number of items dumped
def dump(name, data = [])
processed_data = @json2csv_conversor.process(data, false, @additional_fields[name]) + "\n"
processed_data.encode!(OUTPUT_ENCODING, replace: '')
@files[name].write(processed_data)
@original_files[name].write(::JSON.dump(data) + "\n") if @debug_mode
data.count
end
# @param name String
def end_dump(name)
if @files[name]
@files[name].close
end
if @original_files[name]
@original_files[name].close
end
end
# @param names_list Array
# @param stream IO
def merge_dumps_into_stream(names_list, stream)
headers = @json2csv_conversor.generate_headers(@additional_fields[names_list.first]) + "\n"
streamed_size = headers.length
stream.write(headers)
names_list.each do |name|
input_stream = File.open(@files[name].path)
begin
buffer = input_stream.read(@buffer_size)
if buffer
stream.write(buffer)
streamed_size += buffer.length
end
end while buffer
input_stream.close
@files[name].unlink unless @debug_mode
end
if @debug_mode && !@headers_file.nil?
@headers_file.write(headers)
@headers_file.close
end
streamed_size
end
# @param names_list Array
# @return String
def merge_dumps(names_list = [])
headers = @json2csv_conversor.generate_headers(@additional_fields[names_list.first]) + "\n"
return_data = headers
return_data.encode!(OUTPUT_ENCODING, replace: '')
if @debug_mode && !@headers_file.nil?
@headers_file.write(headers)
@headers_file.close
end
names_list.each do |name|
return_data << File.read(@files[name].path)
@files[name].unlink unless @debug_mode
end
# Remove final trailing newline before returning
return_data.sub(/\n$/, '')
end
# Return a new temporary file contained inside a tmp subfolder
# @param base_name String|nil (optional)
def temporary_file(base_name = '', extension = CONVERTED_FILE_EXTENSION)
FileUtils.mkdir_p(FILE_DUMPER_TMP_SUBFOLDER) unless File.directory?(FILE_DUMPER_TMP_SUBFOLDER)
temps_full_path = FILE_DUMPER_TMP_SUBFOLDER + @temporary_folder + '/'
FileUtils.mkdir_p(temps_full_path)
# For the default scenario force encoding, for original files don't touch anything
if extension == CONVERTED_FILE_EXTENSION
Tempfile.new([base_name.gsub(' ', '_'), extension], temps_full_path, encoding: OUTPUT_ENCODING)
else
Tempfile.new([base_name.gsub(' ', '_'), extension], temps_full_path)
end
end
def file_paths
@files.values.map(&:path)
end
def original_file_paths
@original_files.values.map(&:path)
end
def headers_path
@headers_file.path unless @headers_file.nil?
end
def clean_string(contents)
@json2csv_conversor.clean_string(contents)
end
private
# Intended for tests
def destroy_files
@files.keys.each { |key| @files[key].close! }
@original_files.keys.each { |key| @original_files[key].close! }
@headers_file.close! unless @headers_file.nil?
end
end
end
end
@@ -0,0 +1,79 @@
require 'yaml'
require_relative '../../../../spec/rspec_configuration'
require_relative '../../lib/datasources'
require_relative '../doubles/user'
require 'spec_helper_min'
include CartoDB::Datasources
describe DatasourcesFactory do
def get_config
@config ||= YAML.load_file("#{File.dirname(__FILE__)}/../../../../config/app_config.yml")['defaults']
end
describe '#provider_instantiations' do
it 'tests all available provider instantiations' do
user = FactoryGirl.build(:user)
user.stubs('has_feature_flag?').with('gnip_v2').returns(false)
DatasourcesFactory.set_config(get_config)
dropbox_provider = DatasourcesFactory.get_datasource(Url::Dropbox::DATASOURCE_NAME, user)
dropbox_provider.is_a?(Url::Dropbox).should eq true
dropbox_provider = DatasourcesFactory.get_datasource(Url::Box::DATASOURCE_NAME, user)
dropbox_provider.is_a?(Url::Box).should eq true
# Stubs Google Drive client for connectionless testing
Google::Apis::DriveV2::DriveService.any_instance.stubs(:get_file)
Google::Apis::DriveV2::DriveService.any_instance.stubs(:export_file)
Google::Apis::DriveV2::DriveService.any_instance.stubs(:list_files)
gdrive_provider = DatasourcesFactory.get_datasource(Url::GDrive::DATASOURCE_NAME, user)
gdrive_provider.is_a?(Url::GDrive).should eq true
url_provider = DatasourcesFactory.get_datasource(Url::PublicUrl::DATASOURCE_NAME, user)
url_provider.is_a?(Url::PublicUrl).should eq true
twitter_provider = DatasourcesFactory.get_datasource(Search::Twitter::DATASOURCE_NAME, user)
twitter_provider.is_a?(Search::Twitter).should eq true
nil_provider = DatasourcesFactory.get_datasource(nil, user)
nil_provider.nil?.should eq true
expect {
DatasourcesFactory.get_datasource('blablabla...', user)
}.to raise_exception MissingConfigurationError
end
end
describe '#customized_config?' do
let(:twitter_datasource) { CartoDB::Datasources::Search::Twitter::DATASOURCE_NAME }
before(:each) do
@config = get_config
end
it 'returns false for a random user' do
user = FactoryGirl.build(:carto_user, username: 'wadus')
DatasourcesFactory.customized_config?(twitter_datasource, user).should be_false
end
it 'returns true for a user with custom config' do
user = FactoryGirl.build(:carto_user, username: 'wadus')
@config['datasource_search']['twitter_search']['customized_user_list'] = [user.username]
DatasourcesFactory.set_config(@config)
DatasourcesFactory.customized_config?(twitter_datasource, user).should be_true
end
it 'returns true for a user in an organization with custom config' do
organization = Carto::Organization.new(name: 'wadus-org')
user = FactoryGirl.build(:carto_user, username: 'nowadus', organization: organization)
@config['datasource_search']['twitter_search']['customized_orgs_list'] = [organization.name]
DatasourcesFactory.set_config(@config)
DatasourcesFactory.customized_config?(twitter_datasource, user).should be_true
end
end
end
@@ -0,0 +1,35 @@
require 'yaml'
require_relative '../../lib/datasources'
require_relative '../doubles/user'
include CartoDB::Datasources
describe Url::Dropbox do
def get_config
@config ||= YAML.load_file("#{File.dirname(__FILE__)}/../../../../config/app_config.yml")['defaults']['oauth']['dropbox']
end #get_config
describe '#manual_test' do
it 'with user interaction, tests the full oauth flow and lists files of an account' do
user_mock = CartoDB::Datasources::Doubles::User.new
config = get_config
dropbox_datasource = Url::Dropbox.get_new(config, user_mock)
if config.include?(:access_token)
dropbox_datasource.token = config[:access_token]
else
pending('This test requires manual run, opening the url in a browser, grabbing the code and setting "input" to it')
puts dropbox_datasource.get_auth_url
input = ''
dropbox_datasource.validate_auth_code(input)
puts dropbox_datasource.token
end
data = dropbox_datasource.get_resources_list
puts data
end
end
end
@@ -0,0 +1,36 @@
require 'yaml'
require_relative '../../lib/datasources'
require_relative '../doubles/user'
include CartoDB::Datasources
describe Url::GDrive do
def get_config
@config ||= YAML.load_file("#{File.dirname(__FILE__)}/../../../../config/app_config.yml")['defaults']['oauth']['gdrive']
end
describe '#manual_test' do
it 'with user interaction, tests the full oauth flow and lists files of an account' do
config = get_config
if !config.include?(:refresh_token)
pending('If config unset, this test requires manual running. Check its source code to see what to do')
end
user_mock = CartoDB::Datasources::Doubles::User.new
gdrive_datasource = Url::GDrive.get_new(config, user_mock)
if config.include?(:refresh_token)
gdrive_datasource.token = config[:refresh_token]
else
# Manual testing
puts gdrive_datasource.get_auth_url
input = ''
gdrive_datasource.validate_auth_code(input)
puts gdrive_datasource.token
end
data = gdrive_datasource.get_resources_list
puts data
end
end
end
@@ -0,0 +1,36 @@
require_relative '../../lib/datasources'
require_relative '../../../../spec/helpers/file_server_helper'
include CartoDB::Datasources
include FileServerHelper
describe Url::PublicUrl do
describe '#basic_tests' do
it 'Some basic download flows of this file provider, including error handling' do
url_provider = Url::PublicUrl.get_new
serve_file 'spec/support/data/cartofante_blue.png' do |url|
invalid_url = url + 'invalid'
data = url_provider.get_resource(url)
data.empty?.should eq false
expect {
url_provider.get_resource(invalid_url)
}.to raise_exception DataDownloadError
url_provider.fetch_headers(url)
url_provider.etag_header.empty?.should eq false
url_provider.fetch_headers(invalid_url).should == {}
url_provider.etag_header.should be_empty
url_provider.last_modified_header.should be_empty
# puts data
end
end
end
end
@@ -0,0 +1,20 @@
module CartoDB
module Datasources
module Doubles
class DataImport
attr_accessor :id,
:service_item_id
def initialize(attrs = {})
@id = attrs.fetch(:id, '123456')
@service_item_id = attrs.fetch(:service_item_id, '67890')
end
def save
self
end
end
end
end
end
@@ -0,0 +1,23 @@
module CartoDB
module TwitterSearch
module Doubles
class JSONToCSVConverter
def initialize(attrs = {})
end
def process(input_data = [], add_headers = false, additional_fields = {})
input_data.join("\n")
end
def generate_headers(additional_fields = {})
if additional_fields.nil? || additional_fields.empty?
''
else
additional_fields.join(',')
end
end
end
end
end
end
@@ -0,0 +1,18 @@
module CartoDB
module Datasources
module Doubles
class Organization
attr_accessor :twitter_datasource_enabled
def initialize(attrs = {})
@twitter_datasource_enabled = attrs.fetch(:twitter_datasource_enabled, true)
end
def save
self
end
end
end
end
end
@@ -0,0 +1,26 @@
module CartoDB
module Datasources
module Doubles
class SearchTweet
attr_accessor :user_id,
:data_import_id,
:service_item_id,
:retrieved_items,
:state
def set_importing_state
@state = 'importing'
end
def set_complete_state
@state = 'complete'
end
def save
self
end
end
end
end
end
+45
View File
@@ -0,0 +1,45 @@
module CartoDB
module Datasources
module Doubles
class User
attr_accessor :twitter_datasource_enabled,
:soft_twitter_datasource_limit,
:twitter_datasource_quota,
:username,
:id
def initialize(attrs = {})
@twitter_datasource_enabled = attrs.fetch(:twitter_datasource_enabled, true)
@soft_twitter_datasource_limit = attrs.fetch(:soft_twitter_datasource_limit, false)
@twitter_datasource_quota = attrs.fetch(:twitter_datasource_quota, 123)
@username = attrs.fetch(:username, 'wadus')
@id = attrs.fetch(:id, '000-000')
@organization = attrs.fetch(:has_org, false) \
? Organization.new({
twitter_datasource_enabled: attrs.fetch(:org_twitter_datasource_enabled, true),
twitter_datasource_quota: attrs.fetch(:org_twitter_datasource_quota, 123)
}) \
: nil
end
def organization
@organization
end
def save
self
end
def remaining_twitter_quota
if @organization.nil?
@twitter_datasource_quota
else
@organization.twitter_datasource_quota
end
end
end
end
end
end
+63
View File
@@ -0,0 +1,63 @@
{
"displayFieldName": "NAME",
"fieldAliases": {
"OBJECTID": "OBJECTID",
"WDPAID": "WDPAID",
"NAME": "NAME"
},
"geometryType": "esriGeometryPolygon",
"spatialReference": {
"wkid": 4326,
"latestWkid": 4326
},
"fields": [
{
"name": "OBJECTID",
"type": "esriFieldTypeOID",
"alias": "OBJECTID"
},
{
"name": "WDPAID",
"type": "esriFieldTypeInteger",
"alias": "WDPAID"
},
{
"name": "NAME",
"type": "esriFieldTypeString",
"alias": "NAME",
"length": 254
}
],
"features": [
{
"attributes": {
"OBJECTID": 1,
"WDPAID": 991,
"NAME": "Name of object 1"
},
"geometry": {
"fake": "geom"
}
},
{
"attributes": {
"OBJECTID": 2,
"WDPAID": 992,
"NAME": "Name of object 2"
},
"geometry": {
"fake": "geom"
}
},
{
"attributes": {
"OBJECTID": 3,
"WDPAID": 993,
"NAME": "Name of object 3"
},
"geometry": {
"fake": "geom"
}
}
]
}
@@ -0,0 +1 @@
{"objectIdFieldName":"OBJECTID","objectIds":[1,2,3,4,5,6,7,8,9,10]}
@@ -0,0 +1 @@
{"objectIdFieldName":"OBJECTID","objectIds":[1,2,3]}
+3
View File
@@ -0,0 +1,3 @@
{"layers":[
{"id":0, "name":"first layer", "type":"Feature Layer"}
]}
@@ -0,0 +1,22 @@
{"currentVersion":10.22,
"id":0,
"name":"Test Feature",
"type":"Feature Layer",
"description":"Sample metadata payload",
"geometryType":"esriGeometryPolygon",
"copyrightText":"CartoDB",
"fields":[
{"name":"OBJECTID",
"type":"esriFieldTypeOID",
"alias":"OBJECTID",
"domain":null},
{"name":"NAME",
"type":"esriFieldTypeString",
"alias":"NAME",
"length":254,
"domain":null}
],
"maxRecordCount":1000,
"supportsAdvancedQueries":true,
"supportedQueryFormats":"JSON,AMF",
"useStandardizedQueries":true}
@@ -0,0 +1 @@
{"objectIdFieldName":"OBJECTID","objectIds":[7,6,10,4,5,2,1,8,9,3]}
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,2 @@
"link","body","objectType","postedTime","favoritesCount","twitter_lang","retweetCount","actor_id","actor_displayName","actor_image","actor_summary","actor_postedTime","actor_location","actor_utcOffset","actor_preferredUsername","actor_friendsCount","actor_followersCount","actor_listedCount","actor_statusesCount","actor_verified","inReplyTo_link","geo","twitter_entities","location_geo","location_name","the_geom","category_name","category_terms"
"http://twitter.com/charley_glynn/statuses/834008296361172992","RT @uSIG_CCHS_CSIC: Carto tips: Using blend modes and opacity levels-https://t.co/eGDWGkfWCv via @OrdnanceSurvey https://t.co/qEfFQpLGvL","activity","2017-02-21T11:54:07.000Z","0","en","6","id:twitter.com:492926788","Charley Glynn","https://pbs.twimg.com/profile_images/740123846951460864/gi1ee2lJ_normal.jpg","happy husband, doting dad, Everton fan, music, film & design fan, pushing pixels and joining dots @ordnancesurvey working on #GeoDataViz","2012-02-15T08:23:02.000Z","{""objectType"":""place"",""displayName"":""Southampton, England""}","0","charley_glynn","2482","1415","168","7536","false",,,"{""hashtags"":[],""urls"":[{""url"":""https://t.co/eGDWGkfWCv"",""expanded_url"":""https://www.ordnancesurvey.co.uk/blog/2017/02/carto-tips-using-blend-modes-opacity-levels/"",""display_url"":""ordnancesurvey.co.uk/blog/2017/02/c…"",""indices"":[69,92]}],""user_mentions"":[{""screen_name"":""uSIG_CCHS_CSIC"",""name"":""uSIG (CCHS-CSIC)"",""id"":704613732614344704,""id_str"":""704613732614344704"",""indices"":[3,18]},{""screen_name"":""OrdnanceSurvey"",""name"":""Ordnance Survey"",""id"":22614266,""id_str"":""22614266"",""indices"":[97,112]}],""symbols"":[],""media"":[{""id"":833943995302760449,""id_str"":""833943995302760449"",""indices"":[113,136],""media_url"":""http://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg"",""media_url_https"":""https://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg"",""url"":""https://t.co/qEfFQpLGvL"",""display_url"":""pic.twitter.com/qEfFQpLGvL"",""expanded_url"":""https://twitter.com/uSIG_CCHS_CSIC/status/833972033960620032/photo/1"",""type"":""photo"",""sizes"":{""medium"":{""w"":600,""h"":415,""resize"":""fit""},""large"":{""w"":650,""h"":450,""resize"":""fit""},""thumb"":{""w"":150,""h"":150,""resize"":""crop""},""small"":{""w"":340,""h"":235,""resize"":""fit""}},""source_status_id"":833972033960620032,""source_status_id_str"":""833972033960620032"",""source_user_id"":704613732614344704,""source_user_id_str"":""704613732614344704""}]}",,,"{""coordinates"":[-1.40428,50.90395],""type"":""point""}","1","carto"
1 link body objectType postedTime favoritesCount twitter_lang retweetCount actor_id actor_displayName actor_image actor_summary actor_postedTime actor_location actor_utcOffset actor_preferredUsername actor_friendsCount actor_followersCount actor_listedCount actor_statusesCount actor_verified inReplyTo_link geo twitter_entities location_geo location_name the_geom category_name category_terms
2 http://twitter.com/charley_glynn/statuses/834008296361172992 RT @uSIG_CCHS_CSIC: Carto tips: Using blend modes and opacity levels-https://t.co/eGDWGkfWCv via @OrdnanceSurvey https://t.co/qEfFQpLGvL activity 2017-02-21T11:54:07.000Z 0 en 6 id:twitter.com:492926788 Charley Glynn https://pbs.twimg.com/profile_images/740123846951460864/gi1ee2lJ_normal.jpg happy husband, doting dad, Everton fan, music, film & design fan, pushing pixels and joining dots @ordnancesurvey working on #GeoDataViz 2012-02-15T08:23:02.000Z {"objectType":"place","displayName":"Southampton, England"} 0 charley_glynn 2482 1415 168 7536 false {"hashtags":[],"urls":[{"url":"https://t.co/eGDWGkfWCv","expanded_url":"https://www.ordnancesurvey.co.uk/blog/2017/02/carto-tips-using-blend-modes-opacity-levels/","display_url":"ordnancesurvey.co.uk/blog/2017/02/c…","indices":[69,92]}],"user_mentions":[{"screen_name":"uSIG_CCHS_CSIC","name":"uSIG (CCHS-CSIC)","id":704613732614344704,"id_str":"704613732614344704","indices":[3,18]},{"screen_name":"OrdnanceSurvey","name":"Ordnance Survey","id":22614266,"id_str":"22614266","indices":[97,112]}],"symbols":[],"media":[{"id":833943995302760449,"id_str":"833943995302760449","indices":[113,136],"media_url":"http://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg","media_url_https":"https://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg","url":"https://t.co/qEfFQpLGvL","display_url":"pic.twitter.com/qEfFQpLGvL","expanded_url":"https://twitter.com/uSIG_CCHS_CSIC/status/833972033960620032/photo/1","type":"photo","sizes":{"medium":{"w":600,"h":415,"resize":"fit"},"large":{"w":650,"h":450,"resize":"fit"},"thumb":{"w":150,"h":150,"resize":"crop"},"small":{"w":340,"h":235,"resize":"fit"}},"source_status_id":833972033960620032,"source_status_id_str":"833972033960620032","source_user_id":704613732614344704,"source_user_id_str":"704613732614344704"}]} {"coordinates":[-1.40428,50.90395],"type":"point"} 1 carto
+409
View File
@@ -0,0 +1,409 @@
{
"results": [
{
"id": "tag:search.twitter.com,2005:834008296361172992",
"objectType": "activity",
"verb": "share",
"postedTime": "2017-02-21T11:54:07.000Z",
"generator": {
"displayName": "Twitter Web Client",
"link": "http://twitter.com"
},
"provider": {
"objectType": "service",
"displayName": "Twitter",
"link": "http://www.twitter.com"
},
"link": "http://twitter.com/charley_glynn/statuses/834008296361172992",
"body": "RT @uSIG_CCHS_CSIC: Carto tips: Using blend modes and opacity levels-https://t.co/eGDWGkfWCv via @OrdnanceSurvey https://t.co/qEfFQpLGvL",
"actor": {
"objectType": "person",
"id": "id:twitter.com:492926788",
"link": "http://www.twitter.com/charley_glynn",
"displayName": "Charley Glynn",
"postedTime": "2012-02-15T08:23:02.000Z",
"image": "https://pbs.twimg.com/profile_images/740123846951460864/gi1ee2lJ_normal.jpg",
"summary": "happy husband, doting dad, Everton fan, music, film & design fan, pushing pixels and joining dots @ordnancesurvey working on #GeoDataViz",
"friendsCount": 2482,
"followersCount": 1415,
"listedCount": 168,
"statusesCount": 7536,
"twitterTimeZone": "London",
"verified": false,
"utcOffset": "0",
"preferredUsername": "charley_glynn",
"languages": [
"en"
],
"links": [
{
"href": "http://www.cartoblography.wordpress.com",
"rel": "me"
}
],
"location": {
"objectType": "place",
"displayName": "Southampton, England"
},
"favoritesCount": 7414
},
"object": {
"id": "tag:search.twitter.com,2005:833972033960620032",
"objectType": "activity",
"verb": "post",
"postedTime": "2017-02-21T09:30:02.000Z",
"generator": {
"displayName": "TweetDeck",
"link": "https://about.twitter.com/products/tweetdeck"
},
"provider": {
"objectType": "service",
"displayName": "Twitter",
"link": "http://www.twitter.com"
},
"link": "http://twitter.com/uSIG_CCHS_CSIC/statuses/833972033960620032",
"body": "Carto tips: Using blend modes and opacity levels-https://t.co/eGDWGkfWCv via @OrdnanceSurvey https://t.co/qEfFQpLGvL",
"display_text_range": [
0,
92
],
"actor": {
"objectType": "person",
"id": "id:twitter.com:704613732614344704",
"link": "http://www.twitter.com/uSIG_CCHS_CSIC",
"displayName": "uSIG (CCHS-CSIC)",
"postedTime": "2016-03-01T10:26:19.759Z",
"image": "https://pbs.twimg.com/profile_images/802141753553854464/4ZSz4PI-_normal.jpg",
"summary": "Unidad SIG. Espacio, tiempo, ciencia y tecnología #SIG #Cartografía y #Teledetección en Ciencias Humanas y Sociales @CSIC. #GIS #Cartography #RemoteSensing",
"friendsCount": 317,
"followersCount": 515,
"listedCount": 41,
"statusesCount": 1462,
"twitterTimeZone": null,
"verified": false,
"utcOffset": null,
"preferredUsername": "uSIG_CCHS_CSIC",
"languages": [
"es"
],
"links": [
{
"href": "http://unidadsig.cchs.csic.es/sig/",
"rel": "me"
}
],
"location": {
"objectType": "place",
"displayName": "Madrid, España"
},
"favoritesCount": 811
},
"object": {
"objectType": "note",
"id": "object:search.twitter.com,2005:833972033960620032",
"summary": "Carto tips: Using blend modes and opacity levels-https://t.co/eGDWGkfWCv via @OrdnanceSurvey https://t.co/qEfFQpLGvL",
"link": "http://twitter.com/uSIG_CCHS_CSIC/statuses/833972033960620032",
"postedTime": "2017-02-21T09:30:02.000Z"
},
"favoritesCount": 11,
"twitter_entities": {
"hashtags": [],
"urls": [
{
"url": "https://t.co/eGDWGkfWCv",
"expanded_url": "https://www.ordnancesurvey.co.uk/blog/2017/02/carto-tips-using-blend-modes-opacity-levels/",
"display_url": "ordnancesurvey.co.uk/blog/2017/02/c…",
"indices": [
49,
72
]
}
],
"user_mentions": [
{
"screen_name": "OrdnanceSurvey",
"name": "Ordnance Survey",
"id": 22614266,
"id_str": "22614266",
"indices": [
77,
92
]
}
],
"symbols": [],
"media": [
{
"id": 833943995302760449,
"id_str": "833943995302760449",
"indices": [
93,
116
],
"media_url": "http://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg",
"media_url_https": "https://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg",
"url": "https://t.co/qEfFQpLGvL",
"display_url": "pic.twitter.com/qEfFQpLGvL",
"expanded_url": "https://twitter.com/uSIG_CCHS_CSIC/status/833972033960620032/photo/1",
"type": "photo",
"sizes": {
"medium": {
"w": 600,
"h": 415,
"resize": "fit"
},
"large": {
"w": 650,
"h": 450,
"resize": "fit"
},
"thumb": {
"w": 150,
"h": 150,
"resize": "crop"
},
"small": {
"w": 340,
"h": 235,
"resize": "fit"
}
}
}
]
},
"twitter_extended_entities": {
"media": [
{
"id": 833943995302760449,
"id_str": "833943995302760449",
"indices": [
93,
116
],
"media_url": "http://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg",
"media_url_https": "https://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg",
"url": "https://t.co/qEfFQpLGvL",
"display_url": "pic.twitter.com/qEfFQpLGvL",
"expanded_url": "https://twitter.com/uSIG_CCHS_CSIC/status/833972033960620032/photo/1",
"type": "animated_gif",
"sizes": {
"medium": {
"w": 600,
"h": 415,
"resize": "fit"
},
"large": {
"w": 650,
"h": 450,
"resize": "fit"
},
"thumb": {
"w": 150,
"h": 150,
"resize": "crop"
},
"small": {
"w": 340,
"h": 235,
"resize": "fit"
}
},
"video_info": {
"aspect_ratio": [
13,
9
],
"variants": [
{
"bitrate": 0,
"content_type": "video/mp4",
"url": "https://video.twimg.com/tweet_video/C5LDpTKXUAE1xex.mp4"
}
]
}
}
]
},
"twitter_lang": "en",
"twitter_filter_level": "low"
},
"favoritesCount": 0,
"twitter_entities": {
"hashtags": [],
"urls": [
{
"url": "https://t.co/eGDWGkfWCv",
"expanded_url": "https://www.ordnancesurvey.co.uk/blog/2017/02/carto-tips-using-blend-modes-opacity-levels/",
"display_url": "ordnancesurvey.co.uk/blog/2017/02/c…",
"indices": [
69,
92
]
}
],
"user_mentions": [
{
"screen_name": "uSIG_CCHS_CSIC",
"name": "uSIG (CCHS-CSIC)",
"id": 704613732614344704,
"id_str": "704613732614344704",
"indices": [
3,
18
]
},
{
"screen_name": "OrdnanceSurvey",
"name": "Ordnance Survey",
"id": 22614266,
"id_str": "22614266",
"indices": [
97,
112
]
}
],
"symbols": [],
"media": [
{
"id": 833943995302760449,
"id_str": "833943995302760449",
"indices": [
113,
136
],
"media_url": "http://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg",
"media_url_https": "https://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg",
"url": "https://t.co/qEfFQpLGvL",
"display_url": "pic.twitter.com/qEfFQpLGvL",
"expanded_url": "https://twitter.com/uSIG_CCHS_CSIC/status/833972033960620032/photo/1",
"type": "photo",
"sizes": {
"medium": {
"w": 600,
"h": 415,
"resize": "fit"
},
"large": {
"w": 650,
"h": 450,
"resize": "fit"
},
"thumb": {
"w": 150,
"h": 150,
"resize": "crop"
},
"small": {
"w": 340,
"h": 235,
"resize": "fit"
}
},
"source_status_id": 833972033960620032,
"source_status_id_str": "833972033960620032",
"source_user_id": 704613732614344704,
"source_user_id_str": "704613732614344704"
}
]
},
"twitter_extended_entities": {
"media": [
{
"id": 833943995302760449,
"id_str": "833943995302760449",
"indices": [
113,
136
],
"media_url": "http://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg",
"media_url_https": "https://pbs.twimg.com/tweet_video_thumb/C5LDpTKXUAE1xex.jpg",
"url": "https://t.co/qEfFQpLGvL",
"display_url": "pic.twitter.com/qEfFQpLGvL",
"expanded_url": "https://twitter.com/uSIG_CCHS_CSIC/status/833972033960620032/photo/1",
"type": "animated_gif",
"sizes": {
"medium": {
"w": 600,
"h": 415,
"resize": "fit"
},
"large": {
"w": 650,
"h": 450,
"resize": "fit"
},
"thumb": {
"w": 150,
"h": 150,
"resize": "crop"
},
"small": {
"w": 340,
"h": 235,
"resize": "fit"
}
},
"source_status_id": 833972033960620032,
"source_status_id_str": "833972033960620032",
"source_user_id": 704613732614344704,
"source_user_id_str": "704613732614344704",
"video_info": {
"aspect_ratio": [
13,
9
],
"variants": [
{
"bitrate": 0,
"content_type": "video/mp4",
"url": "https://video.twimg.com/tweet_video/C5LDpTKXUAE1xex.mp4"
}
]
}
}
]
},
"twitter_lang": "en",
"retweetCount": 6,
"gnip": {
"profileLocations": [
{
"address": {
"country": "United Kingdom",
"countryCode": "GB",
"locality": "Southampton",
"region": "England",
"subRegion": "City of Southampton"
},
"displayName": "Southampton, England, United Kingdom",
"geo": {
"coordinates": [
-1.40428,
50.90395
],
"type": "point"
},
"objectType": "place"
}
],
"matching_rules": [
{
"value": "(carto) (has:geo OR has:profile_geo)",
"tag": null
}
],
"urls": [
{
"url": "https://t.co/eGDWGkfWCv",
"expanded_url": "https://www.ordnancesurvey.co.uk/blog/2017/02/carto-tips-using-blend-modes-opacity-levels/",
"expanded_status": 200,
"expanded_url_title": "Carto tips: Using blend modes and opacity levels - Ordnance Survey Blog",
"expanded_url_description": "Colour is one of the main graphic elements that a cartographer uses to make their map clear to read. Amongst other things we use colour to create familiarity, to differentiate features and to create a clear visual hierarchy. There are many things we can do to the features on our maps to change their appearance... Read More"
}
]
},
"twitter_filter_level": "low"
}
]
}
+1
View File
@@ -0,0 +1 @@
stream_input_1
+1
View File
@@ -0,0 +1 @@
stream_input_2
@@ -0,0 +1,40 @@
"id","verb","link","body","objectType","postedTime","favoritesCount","twitter_filter_level","twitter_lang","retweetCount","actor_objectType","actor_id","actor_link","actor_displayName","actor_image","actor_summary","actor_postedTime","actor_links","actor_location","actor_utcOffset","actor_preferredUsername","actor_languages","actor_twitterTimeZone","actor_friendsCount","actor_followersCount","actor_listedCount","actor_statusesCount","actor_verified","generator_displayName","generator_link","provider_objectType","provider_displayName","provider_link","inReplyTo_link","geo","twitter_entities","object_objectType","object_id","object_summary","object_postedTime","object_link","location_objectType","location_displayName","location_link","location_geo","location_streetAddress","location_name","gnip","the_geom","category_name","category_terms"
"tag:search.twitter.com,2005:496274441584668672","post","http://twitter.com/erictheise/statuses/496274441584668672","sad to be the guy that knocked ""locahost"" out of @cartoDB's docs for #Odyssey.js; i've certainly had that sentiment towards my dev machine.","activity","2014-08-04T12:40:21.000Z","0","medium","en","0","person","id:twitter.com:94817737","http://www.twitter.com/erictheise","Eric Theise","https://pbs.twimg.com/profile_images/1994194116/Theise_normal.jpg","Software engineer; photographer, experimental filmmaker, cartographer, geographer, vocalizer, yoga student, reader & writer, eater & drinker.","2009-12-05T16:06:12.000Z","[{""href"":null,""rel"":""me""}]","{""objectType"":""place"",""displayName"":""San Francisco""}","-25200","erictheise","[""en""]","Pacific Time (US & Canada)","698","306","25","1322","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[{""text"":""Odyssey"",""indices"":[69,77]}],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[49,57]}],""symbols"":[]}","note","object:search.twitter.com,2005:496274441584668672","sad to be the guy that knocked ""locahost"" out of @cartoDB's docs for #Odyssey.js; i've certainly had that sentiment towards my dev machine.","2014-08-04T12:40:21.000Z","http://twitter.com/erictheise/statuses/496274441584668672","place","San Francisco, CA","https://api.twitter.com/1.1/geo/id/5a110d312052166f.json","{""type"":""Polygon"",""coordinates"":[[[-122.514926,37.708075],[-122.514926,37.833238],[-122.328001,37.833238],[-122.328001,37.708075]]]}",,"San Francisco","{""klout_score"":41,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-122.514926,37.708075],[-122.514926,37.833238],[-122.328001,37.833238],[-122.328001,37.708075]]]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:495565956908064769","post","http://twitter.com/Jmholleran/statuses/495565956908064769","FrackMap #map http://t.co/BLWl3zYKTn","activity","2014-08-02T13:45:05.000Z","0","medium","en","0","person","id:twitter.com:27128139","http://www.twitter.com/Jmholleran","John Munro Holleran","https://pbs.twimg.com/profile_images/460662090819051520/lzLtBzUb_normal.png","Modest! intelligent, articulate & generally just amazing! Trade Unionist, learning facilitator 'n that","2009-03-27T23:36:39.000Z","[{""href"":null,""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Glasgow/Scotland""}",,"Jmholleran","[""en""]",,"411","192","1","1643","false","Twitter for Android","http://twitter.com/download/android","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[55.8641,-4.282299]}","{""hashtags"":[{""text"":""map"",""indices"":[9,13]}],""trends"":[],""urls"":[{""url"":""http://t.co/BLWl3zYKTn"",""expanded_url"":""http://energydesk.cartodb.com/viz/18162194-ecbd-11e3-aba1-0e230854a1cb/embed_map?title=true&description=true&search=false&shareable=true&cartodb_logo=true&layer_selector=false&legends=true&scrollwheel=true&fullscreen=true&sublayer_options=1%7C1%7C1%7C1&sql=&sw_lat=49.97948776108648&sw_lon=-12.7001953125&ne_lat=56.80087831233043&ne_lon=8.96484375"",""display_url"":""energydesk.cartodb.com/viz/18162194-e…"",""indices"":[14,36]}],""user_mentions"":[],""symbols"":[]}","note","object:search.twitter.com,2005:495565956908064769","FrackMap #map http://t.co/BLWl3zYKTn","2014-08-02T13:45:05.000Z","http://twitter.com/Jmholleran/statuses/495565956908064769","place","Glasgow","https://api.twitter.com/1.1/geo/id/791e00bcadc4615f.json","{""type"":""Polygon"",""coordinates"":[[[-4.3932845,55.796184],[-4.3932845,55.9204214],[-4.0902182,55.9204214],[-4.0902182,55.796184]]]}",,"Glasgow","{""urls"":[{""url"":""http://t.co/BLWl3zYKTn"",""expanded_url"":""http://energydesk.cartodb.com/viz/18162194-ecbd-11e3-aba1-0e230854a1cb/embed_map?title=true&description=true&search=false&shareable=true&cartodb_logo=true&layer_selector=false&legends=true&scrollwheel=true&fullscreen=true&sublayer_options=1%7C1%7C1%7C1&sql&sw_lat=49.97948776108648&sw_lon=-12.7001953125&ne_lat=56.80087831233043&ne_lon=8.96484375"",""expanded_status"":200}],""klout_score"":23}","{""type"":""Point"",""coordinates"":[-4.282299,55.8641]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:494521462682685440","post","http://twitter.com/SonOfJorEl/statuses/494521462682685440","@iriberri1 @cartoDB you are saying it is not supported, but you have an entire blog post about said feature : http://t.co/gvuUWv9sge","activity","2014-07-30T16:34:39.000Z","0","medium","en","0","person","id:twitter.com:15628310","http://www.twitter.com/SonOfJorEl","Joey Schluchter","https://pbs.twimg.com/profile_images/466697750013562880/ngwKpO1j_normal.jpeg","Recovering golfer. Helping feed the world to software. Still love persimmon.","2008-07-28T05:34:03.000Z","[{""href"":""http://www.exzeo.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Clearwater, FL""}","-14400","SonOfJorEl","[""en""]","Eastern Time (US & Canada)","368","2053","5","2143","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com","http://twitter.com/iriberri1/statuses/494488277765091328",,"{""hashtags"":[],""symbols"":[],""urls"":[{""url"":""http://t.co/gvuUWv9sge"",""expanded_url"":""http://blog.cartodb.com/got-files-weve-got-a-import-api/"",""display_url"":""blog.cartodb.com/got-files-weve…"",""indices"":[110,132]}],""user_mentions"":[{""screen_name"":""iriberri1"",""name"":""Carla"",""id"":102197411,""id_str"":""102197411"",""indices"":[0,10]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[11,19]}]}","note","object:search.twitter.com,2005:494521462682685440","@iriberri1 @cartoDB you are saying it is not supported, but you have an entire blog post about said feature : http://t.co/gvuUWv9sge","2014-07-30T16:34:39.000Z","http://twitter.com/SonOfJorEl/statuses/494521462682685440","place","Tampa, FL","https://api.twitter.com/1.1/geo/id/dc62519fda13b4ec.json","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}",,"Tampa","{""urls"":[{""url"":""http://t.co/gvuUWv9sge"",""expanded_url"":""http://blog.cartodb.com/got-files-weve-got-a-import-api/"",""expanded_status"":200}],""klout_score"":42,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:494511938563342336","post","http://twitter.com/xavijam/statuses/494511938563342336",". @Javier is giving birth last @cartodb feature. All I can say is … GIF over maps! http://t.co/oC3DEK5cP6","activity","2014-07-30T15:56:48.000Z","0","medium","en","0","person","id:twitter.com:26775253","http://www.twitter.com/xavijam","Javier Álvarez","https://pbs.twimg.com/profile_images/3025038956/a2919d353eb2b1a22756d7ac79847480_normal.jpeg","I love gentoo penguins","2009-03-26T15:36:52.000Z","[{""href"":""http://xavij.am"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Antarctic Peninsula""}","7200","xavijam","[""en""]","Madrid","250","395","26","4981","false","Tweetbot for iΟS","http://tapbots.com/tweetbot","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""javier"",""name"":""Javier Arce"",""id"":39083,""id_str"":""39083"",""indices"":[2,9]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[31,39]}],""symbols"":[],""media"":[{""id"":494511934629093378,""id_str"":""494511934629093378"",""indices"":[83,105],""media_url"":""http://pbs.twimg.com/media/Btzb_AjCIAI0cLr.jpg"",""media_url_https"":""https://pbs.twimg.com/media/Btzb_AjCIAI0cLr.jpg"",""url"":""http://t.co/oC3DEK5cP6"",""display_url"":""pic.twitter.com/oC3DEK5cP6"",""expanded_url"":""http://twitter.com/xavijam/status/494511938563342336/photo/1"",""type"":""photo"",""sizes"":{""small"":{""w"":340,""h"":255,""resize"":""fit""},""medium"":{""w"":600,""h"":450,""resize"":""fit""},""large"":{""w"":1024,""h"":768,""resize"":""fit""},""thumb"":{""w"":150,""h"":150,""resize"":""crop""}}}]}","note","object:search.twitter.com,2005:494511938563342336",". @Javier is giving birth last @cartodb feature. All I can say is … GIF over maps! http://t.co/oC3DEK5cP6","2014-07-30T15:56:48.000Z","http://twitter.com/xavijam/statuses/494511938563342336","place","Trafalgar","https://api.twitter.com/1.1/geo/id/0144b1172069289c.json","{""type"":""Polygon"",""coordinates"":[[[-3.70582,40.4275708],[-3.70582,40.4386747],[-3.6956326,40.4386747],[-3.6956326,40.4275708]]]}",,"Trafalgar","{""urls"":[{""url"":""http://t.co/oC3DEK5cP6"",""expanded_url"":""http://twitter.com/xavijam/status/494511938563342336/photo/1"",""expanded_status"":200}],""klout_score"":39,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-3.70582,40.4275708],[-3.70582,40.4386747],[-3.6956326,40.4386747],[-3.6956326,40.4275708]]]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:494485754848894978","post","http://twitter.com/SonOfJorEl/statuses/494485754848894978","@cartoDB I’ve emailed support (which you claim has a 24hr turnaround) with minimal response. I am a paying customer ($149 mo). For how long?","activity","2014-07-30T14:12:45.000Z","0","medium","en","0","person","id:twitter.com:15628310","http://www.twitter.com/SonOfJorEl","Joey Schluchter","https://pbs.twimg.com/profile_images/466697750013562880/ngwKpO1j_normal.jpeg","Recovering golfer. Helping feed the world to software. Still love persimmon.","2008-07-28T05:34:03.000Z","[{""href"":""http://www.exzeo.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Clearwater, FL""}","-14400","SonOfJorEl","[""en""]","Eastern Time (US & Canada)","368","2053","5","2142","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[0,8]}],""symbols"":[]}","note","object:search.twitter.com,2005:494485754848894978","@cartoDB I’ve emailed support (which you claim has a 24hr turnaround) with minimal response. I am a paying customer ($149 mo). For how long?","2014-07-30T14:12:45.000Z","http://twitter.com/SonOfJorEl/statuses/494485754848894978","place","Tampa, FL","https://api.twitter.com/1.1/geo/id/dc62519fda13b4ec.json","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}",,"Tampa","{""klout_score"":42,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:494276921765543936","post","http://twitter.com/httsan/statuses/494276921765543936","""@gisuser: Online Mapping for Beginners – The Map Academy from CartoDB: TweetSomething innovative from… http://t.co/Zhh882TV2F""","activity","2014-07-30T00:22:56.000Z","0","medium","en","0","person","id:twitter.com:28471178","http://www.twitter.com/httsan","hartanto","https://pbs.twimg.com/profile_images/2217327231/IMG0033_normal.jpg","Lahir di Indonesia | Nguli di @NEOnetBPPT @BPPTeknologi | Ngobrol di @RSGISForum @PadangZaitun | Nyantri di @MichiganStateU","2009-04-03T01:32:35.000Z","[{""href"":""http://about.me/httsan"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""East Lansing""}","-14400","httsan","[""en""]","Eastern Time (US & Canada)","976","873","12","25725","false","Twitter for Android","http://twitter.com/download/android","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[42.7539688,-84.4221123]}","{""hashtags"":[],""symbols"":[],""urls"":[{""url"":""http://t.co/Zhh882TV2F"",""expanded_url"":""http://goo.gl/fb/xVB5rr"",""display_url"":""goo.gl/fb/xVB5rr"",""indices"":[104,126]}],""user_mentions"":[{""screen_name"":""gisuser"",""name"":""GISuser GeoTech News"",""id"":16958615,""id_str"":""16958615"",""indices"":[1,9]}]}","note","object:search.twitter.com,2005:494276921765543936","""@gisuser: Online Mapping for Beginners – The Map Academy from CartoDB: TweetSomething innovative from… http://t.co/Zhh882TV2F""","2014-07-30T00:22:56.000Z","http://twitter.com/httsan/statuses/494276921765543936","place","Haslett, MI","https://api.twitter.com/1.1/geo/id/4399b5004a3b4d9a.json","{""type"":""Polygon"",""coordinates"":[[[-84.447506,42.731229],[-84.447506,42.769688],[-84.363432,42.769688],[-84.363432,42.731229]]]}",,"Haslett","{""urls"":[{""url"":""http://t.co/Zhh882TV2F"",""expanded_url"":""http://blog.gisuser.com/2014/07/29/online-mapping-for-beginners-the-map-academy-from-cartodb/?utm_source=feedburner&utm_medium=twitter&utm_campaign=Feed:%20gisuser%20(GISUser.com%20-%20GIS,%20Mapping,%20Geospatial,%20and%20location%20technology%20news)"",""expanded_status"":200}],""klout_score"":51,""language"":{""value"":""en""}}","{""type"":""Point"",""coordinates"":[-84.4221123,42.7539688]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:494274906528288769","post","http://twitter.com/carygeo/statuses/494274906528288769","Attention mappers: iframe, #javascript, app development examples from the #CartoDB API/team! Well done @andrewxhill. http://t.co/cJM5fzgoyL","activity","2014-07-30T00:14:55.000Z","0","medium","en","0","person","id:twitter.com:2641688154","http://www.twitter.com/carygeo","Cary Greenwood","https://pbs.twimg.com/profile_images/488501325697544192/gFMxHuLV_normal.jpeg","geography, maps, data visualization, ui/ux design","2014-07-14T01:35:22.000Z","[{""href"":null,""rel"":""me""}]",,,"carygeo","[""en""]",,"109","46","0","47","false","Twitter for iPhone","http://twitter.com/download/iphone","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[34.13846767,-118.3620715]}","{""hashtags"":[{""text"":""javascript"",""indices"":[27,38]},{""text"":""CartoDB"",""indices"":[74,82]}],""symbols"":[],""urls"":[{""url"":""http://t.co/cJM5fzgoyL"",""expanded_url"":""http://vimeo.com/m/101936210"",""display_url"":""vimeo.com/m/101936210"",""indices"":[117,139]}],""user_mentions"":[{""screen_name"":""andrewxhill"",""name"":""Andrew W Hill"",""id"":19893224,""id_str"":""19893224"",""indices"":[103,115]}]}","note","object:search.twitter.com,2005:494274906528288769","Attention mappers: iframe, #javascript, app development examples from the #CartoDB API/team! Well done @andrewxhill. http://t.co/cJM5fzgoyL","2014-07-30T00:14:55.000Z","http://twitter.com/carygeo/statuses/494274906528288769","place","Los Angeles, CA","https://api.twitter.com/1.1/geo/id/3b77caf94bfc81fe.json","{""type"":""Polygon"",""coordinates"":[[[-118.668404,33.704538],[-118.668404,34.330724],[-118.155409,34.330724],[-118.155409,33.704538]]]}",,"Los Angeles","{""urls"":[{""url"":""http://t.co/cJM5fzgoyL"",""expanded_url"":""http://vimeo.com/101936210"",""expanded_status"":200}],""klout_score"":31,""language"":{""value"":""en""}}","{""type"":""Point"",""coordinates"":[-118.3620715,34.13846767]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:494194150133481472","post","http://twitter.com/juanjeojeda/statuses/494194150133481472","@dig_geo_com Your welcome :-) Keep in mind that some parts of @cartoDB still doesn't work fine. We're waiting for update from our provider.","activity","2014-07-29T18:54:01.000Z","0","medium","en","0","person","id:twitter.com:12679112","http://www.twitter.com/juanjeojeda","Juanje Ojeda ","https://pbs.twimg.com/profile_images/46025702/juanje_normal.jpg","Canarian guy who likes the open knowledge, the freedom and learn things every day. #OpenSource #DevOps #PaaS #Canarias","2008-01-25T11:49:15.000Z","[{""href"":""http://about.me/juanje"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Las Palmas de Gran Canaria""}","3600","juanjeojeda","[""es""]","London","416","1006","86","17700","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com","http://twitter.com/dig_geo_com/statuses/494191759078211584",,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""dig_geo_com"",""name"":""digital-geography"",""id"":231016902,""id_str"":""231016902"",""indices"":[0,12]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[62,70]}],""symbols"":[]}","note","object:search.twitter.com,2005:494194150133481472","@dig_geo_com Your welcome :-) Keep in mind that some parts of @cartoDB still doesn't work fine. We're waiting for update from our provider.","2014-07-29T18:54:01.000Z","http://twitter.com/juanjeojeda/statuses/494194150133481472","place","Las Palmas de Gran Canaria, Las Palmas","https://api.twitter.com/1.1/geo/id/00ab08eef1b62e92.json","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}",,"Las Palmas de Gran Canaria","{""klout_score"":56,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:494191583240388608","post","http://twitter.com/juanjeojeda/statuses/494191583240388608","@dig_geo_com Hi, @cartoDB blog is already back. We had some issues with our hosting provider. Sorry for the inconveniences.","activity","2014-07-29T18:43:49.000Z","0","medium","en","0","person","id:twitter.com:12679112","http://www.twitter.com/juanjeojeda","Juanje Ojeda ","https://pbs.twimg.com/profile_images/46025702/juanje_normal.jpg","Canarian guy who likes the open knowledge, the freedom and learn things every day. #OpenSource #DevOps #PaaS #Canarias","2008-01-25T11:49:15.000Z","[{""href"":""http://about.me/juanje"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Las Palmas de Gran Canaria""}","3600","juanjeojeda","[""es""]","London","416","1006","86","17698","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com","http://twitter.com/dig_geo_com/statuses/494181592685117442",,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""dig_geo_com"",""name"":""digital-geography"",""id"":231016902,""id_str"":""231016902"",""indices"":[0,12]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[17,25]}],""symbols"":[]}","note","object:search.twitter.com,2005:494191583240388608","@dig_geo_com Hi, @cartoDB blog is already back. We had some issues with our hosting provider. Sorry for the inconveniences.","2014-07-29T18:43:49.000Z","http://twitter.com/juanjeojeda/statuses/494191583240388608","place","Las Palmas de Gran Canaria, Las Palmas","https://api.twitter.com/1.1/geo/id/00ab08eef1b62e92.json","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}",,"Las Palmas de Gran Canaria","{""klout_score"":56,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:494141648922611712","post","http://twitter.com/spara/statuses/494141648922611712","@vtcraghead @briantimoney @billdollins CartoDB needs to cut a deal with a cloud provider, they’re a bit too parsimonious on disk space","activity","2014-07-29T15:25:24.000Z","0","medium","en","0","person","id:twitter.com:14629939","http://www.twitter.com/spara","spara","https://pbs.twimg.com/profile_images/469468370413166592/uuOHhLby_normal.jpeg","Ut mitterent eos in faciem.","2008-05-02T19:24:50.000Z","[{""href"":""http://sproke.blogspot.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""San Antonio Francisco""}","-25200","spara","[""en""]","Pacific Time (US & Canada)","1311","1276","128","24585","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com","http://twitter.com/vtcraghead/statuses/494140334226833408",,"{""hashtags"":[],""symbols"":[],""urls"":[],""user_mentions"":[{""screen_name"":""vtcraghead"",""name"":""Bill Morris"",""id"":278873782,""id_str"":""278873782"",""indices"":[0,11]},{""screen_name"":""briantimoney"",""name"":""Brian Timoney"",""id"":17546328,""id_str"":""17546328"",""indices"":[12,25]},{""screen_name"":""billdollins"",""name"":""Bill Dollins"",""id"":12405802,""id_str"":""12405802"",""indices"":[26,38]}]}","note","object:search.twitter.com,2005:494141648922611712","@vtcraghead @briantimoney @billdollins CartoDB needs to cut a deal with a cloud provider, they’re a bit too parsimonious on disk space","2014-07-29T15:25:24.000Z","http://twitter.com/spara/statuses/494141648922611712","place","Highland Park, San Antonio","https://api.twitter.com/1.1/geo/id/097c1754d9aa7b39.json","{""type"":""Polygon"",""coordinates"":[[[-98.478227,29.384112],[-98.478227,29.401578],[-98.448015,29.401578],[-98.448015,29.384112]]]}",,"Highland Park","{""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-98.478227,29.384112],[-98.478227,29.401578],[-98.448015,29.401578],[-98.448015,29.384112]]]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:496274441584668672","post","http://twitter.com/erictheise/statuses/496274441584668672","sad to be the guy that knocked ""locahost"" out of @cartoDB's docs for #Odyssey.js; i've certainly had that sentiment towards my dev machine.","activity","2014-08-04T12:40:21.000Z","0","medium","en","0","person","id:twitter.com:94817737","http://www.twitter.com/erictheise","Eric Theise","https://pbs.twimg.com/profile_images/1994194116/Theise_normal.jpg","Software engineer; photographer, experimental filmmaker, cartographer, geographer, vocalizer, yoga student, reader & writer, eater & drinker.","2009-12-05T16:06:12.000Z","[{""href"":null,""rel"":""me""}]","{""objectType"":""place"",""displayName"":""San Francisco""}","-25200","erictheise","[""en""]","Pacific Time (US & Canada)","698","306","25","1322","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[{""text"":""Odyssey"",""indices"":[69,77]}],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[49,57]}],""symbols"":[]}","note","object:search.twitter.com,2005:496274441584668672","sad to be the guy that knocked ""locahost"" out of @cartoDB's docs for #Odyssey.js; i've certainly had that sentiment towards my dev machine.","2014-08-04T12:40:21.000Z","http://twitter.com/erictheise/statuses/496274441584668672","place","San Francisco, CA","https://api.twitter.com/1.1/geo/id/5a110d312052166f.json","{""type"":""Polygon"",""coordinates"":[[[-122.514926,37.708075],[-122.514926,37.833238],[-122.328001,37.833238],[-122.328001,37.708075]]]}",,"San Francisco","{""klout_score"":41,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-122.514926,37.708075],[-122.514926,37.833238],[-122.328001,37.833238],[-122.328001,37.708075]]]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:495565956908064769","post","http://twitter.com/Jmholleran/statuses/495565956908064769","FrackMap #map http://t.co/BLWl3zYKTn","activity","2014-08-02T13:45:05.000Z","0","medium","en","0","person","id:twitter.com:27128139","http://www.twitter.com/Jmholleran","John Munro Holleran","https://pbs.twimg.com/profile_images/460662090819051520/lzLtBzUb_normal.png","Modest! intelligent, articulate & generally just amazing! Trade Unionist, learning facilitator 'n that","2009-03-27T23:36:39.000Z","[{""href"":null,""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Glasgow/Scotland""}",,"Jmholleran","[""en""]",,"411","192","1","1643","false","Twitter for Android","http://twitter.com/download/android","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[55.8641,-4.282299]}","{""hashtags"":[{""text"":""map"",""indices"":[9,13]}],""trends"":[],""urls"":[{""url"":""http://t.co/BLWl3zYKTn"",""expanded_url"":""http://energydesk.cartodb.com/viz/18162194-ecbd-11e3-aba1-0e230854a1cb/embed_map?title=true&description=true&search=false&shareable=true&cartodb_logo=true&layer_selector=false&legends=true&scrollwheel=true&fullscreen=true&sublayer_options=1%7C1%7C1%7C1&sql=&sw_lat=49.97948776108648&sw_lon=-12.7001953125&ne_lat=56.80087831233043&ne_lon=8.96484375"",""display_url"":""energydesk.cartodb.com/viz/18162194-e…"",""indices"":[14,36]}],""user_mentions"":[],""symbols"":[]}","note","object:search.twitter.com,2005:495565956908064769","FrackMap #map http://t.co/BLWl3zYKTn","2014-08-02T13:45:05.000Z","http://twitter.com/Jmholleran/statuses/495565956908064769","place","Glasgow","https://api.twitter.com/1.1/geo/id/791e00bcadc4615f.json","{""type"":""Polygon"",""coordinates"":[[[-4.3932845,55.796184],[-4.3932845,55.9204214],[-4.0902182,55.9204214],[-4.0902182,55.796184]]]}",,"Glasgow","{""urls"":[{""url"":""http://t.co/BLWl3zYKTn"",""expanded_url"":""http://energydesk.cartodb.com/viz/18162194-ecbd-11e3-aba1-0e230854a1cb/embed_map?title=true&description=true&search=false&shareable=true&cartodb_logo=true&layer_selector=false&legends=true&scrollwheel=true&fullscreen=true&sublayer_options=1%7C1%7C1%7C1&sql&sw_lat=49.97948776108648&sw_lon=-12.7001953125&ne_lat=56.80087831233043&ne_lon=8.96484375"",""expanded_status"":200}],""klout_score"":23}","{""type"":""Point"",""coordinates"":[-4.282299,55.8641]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:494521462682685440","post","http://twitter.com/SonOfJorEl/statuses/494521462682685440","@iriberri1 @cartoDB you are saying it is not supported, but you have an entire blog post about said feature : http://t.co/gvuUWv9sge","activity","2014-07-30T16:34:39.000Z","0","medium","en","0","person","id:twitter.com:15628310","http://www.twitter.com/SonOfJorEl","Joey Schluchter","https://pbs.twimg.com/profile_images/466697750013562880/ngwKpO1j_normal.jpeg","Recovering golfer. Helping feed the world to software. Still love persimmon.","2008-07-28T05:34:03.000Z","[{""href"":""http://www.exzeo.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Clearwater, FL""}","-14400","SonOfJorEl","[""en""]","Eastern Time (US & Canada)","368","2053","5","2143","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com","http://twitter.com/iriberri1/statuses/494488277765091328",,"{""hashtags"":[],""symbols"":[],""urls"":[{""url"":""http://t.co/gvuUWv9sge"",""expanded_url"":""http://blog.cartodb.com/got-files-weve-got-a-import-api/"",""display_url"":""blog.cartodb.com/got-files-weve…"",""indices"":[110,132]}],""user_mentions"":[{""screen_name"":""iriberri1"",""name"":""Carla"",""id"":102197411,""id_str"":""102197411"",""indices"":[0,10]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[11,19]}]}","note","object:search.twitter.com,2005:494521462682685440","@iriberri1 @cartoDB you are saying it is not supported, but you have an entire blog post about said feature : http://t.co/gvuUWv9sge","2014-07-30T16:34:39.000Z","http://twitter.com/SonOfJorEl/statuses/494521462682685440","place","Tampa, FL","https://api.twitter.com/1.1/geo/id/dc62519fda13b4ec.json","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}",,"Tampa","{""urls"":[{""url"":""http://t.co/gvuUWv9sge"",""expanded_url"":""http://blog.cartodb.com/got-files-weve-got-a-import-api/"",""expanded_status"":200}],""klout_score"":42,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:494511938563342336","post","http://twitter.com/xavijam/statuses/494511938563342336",". @Javier is giving birth last @cartodb feature. All I can say is … GIF over maps! http://t.co/oC3DEK5cP6","activity","2014-07-30T15:56:48.000Z","0","medium","en","0","person","id:twitter.com:26775253","http://www.twitter.com/xavijam","Javier Álvarez","https://pbs.twimg.com/profile_images/3025038956/a2919d353eb2b1a22756d7ac79847480_normal.jpeg","I love gentoo penguins","2009-03-26T15:36:52.000Z","[{""href"":""http://xavij.am"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Antarctic Peninsula""}","7200","xavijam","[""en""]","Madrid","250","395","26","4981","false","Tweetbot for iΟS","http://tapbots.com/tweetbot","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""javier"",""name"":""Javier Arce"",""id"":39083,""id_str"":""39083"",""indices"":[2,9]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[31,39]}],""symbols"":[],""media"":[{""id"":494511934629093378,""id_str"":""494511934629093378"",""indices"":[83,105],""media_url"":""http://pbs.twimg.com/media/Btzb_AjCIAI0cLr.jpg"",""media_url_https"":""https://pbs.twimg.com/media/Btzb_AjCIAI0cLr.jpg"",""url"":""http://t.co/oC3DEK5cP6"",""display_url"":""pic.twitter.com/oC3DEK5cP6"",""expanded_url"":""http://twitter.com/xavijam/status/494511938563342336/photo/1"",""type"":""photo"",""sizes"":{""small"":{""w"":340,""h"":255,""resize"":""fit""},""medium"":{""w"":600,""h"":450,""resize"":""fit""},""large"":{""w"":1024,""h"":768,""resize"":""fit""},""thumb"":{""w"":150,""h"":150,""resize"":""crop""}}}]}","note","object:search.twitter.com,2005:494511938563342336",". @Javier is giving birth last @cartodb feature. All I can say is … GIF over maps! http://t.co/oC3DEK5cP6","2014-07-30T15:56:48.000Z","http://twitter.com/xavijam/statuses/494511938563342336","place","Trafalgar","https://api.twitter.com/1.1/geo/id/0144b1172069289c.json","{""type"":""Polygon"",""coordinates"":[[[-3.70582,40.4275708],[-3.70582,40.4386747],[-3.6956326,40.4386747],[-3.6956326,40.4275708]]]}",,"Trafalgar","{""urls"":[{""url"":""http://t.co/oC3DEK5cP6"",""expanded_url"":""http://twitter.com/xavijam/status/494511938563342336/photo/1"",""expanded_status"":200}],""klout_score"":39,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-3.70582,40.4275708],[-3.70582,40.4386747],[-3.6956326,40.4386747],[-3.6956326,40.4275708]]]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:494485754848894978","post","http://twitter.com/SonOfJorEl/statuses/494485754848894978","@cartoDB I’ve emailed support (which you claim has a 24hr turnaround) with minimal response. I am a paying customer ($149 mo). For how long?","activity","2014-07-30T14:12:45.000Z","0","medium","en","0","person","id:twitter.com:15628310","http://www.twitter.com/SonOfJorEl","Joey Schluchter","https://pbs.twimg.com/profile_images/466697750013562880/ngwKpO1j_normal.jpeg","Recovering golfer. Helping feed the world to software. Still love persimmon.","2008-07-28T05:34:03.000Z","[{""href"":""http://www.exzeo.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Clearwater, FL""}","-14400","SonOfJorEl","[""en""]","Eastern Time (US & Canada)","368","2053","5","2142","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[0,8]}],""symbols"":[]}","note","object:search.twitter.com,2005:494485754848894978","@cartoDB I’ve emailed support (which you claim has a 24hr turnaround) with minimal response. I am a paying customer ($149 mo). For how long?","2014-07-30T14:12:45.000Z","http://twitter.com/SonOfJorEl/statuses/494485754848894978","place","Tampa, FL","https://api.twitter.com/1.1/geo/id/dc62519fda13b4ec.json","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}",,"Tampa","{""klout_score"":42,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:494276921765543936","post","http://twitter.com/httsan/statuses/494276921765543936","""@gisuser: Online Mapping for Beginners – The Map Academy from CartoDB: TweetSomething innovative from… http://t.co/Zhh882TV2F""","activity","2014-07-30T00:22:56.000Z","0","medium","en","0","person","id:twitter.com:28471178","http://www.twitter.com/httsan","hartanto","https://pbs.twimg.com/profile_images/2217327231/IMG0033_normal.jpg","Lahir di Indonesia | Nguli di @NEOnetBPPT @BPPTeknologi | Ngobrol di @RSGISForum @PadangZaitun | Nyantri di @MichiganStateU","2009-04-03T01:32:35.000Z","[{""href"":""http://about.me/httsan"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""East Lansing""}","-14400","httsan","[""en""]","Eastern Time (US & Canada)","976","873","12","25725","false","Twitter for Android","http://twitter.com/download/android","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[42.7539688,-84.4221123]}","{""hashtags"":[],""symbols"":[],""urls"":[{""url"":""http://t.co/Zhh882TV2F"",""expanded_url"":""http://goo.gl/fb/xVB5rr"",""display_url"":""goo.gl/fb/xVB5rr"",""indices"":[104,126]}],""user_mentions"":[{""screen_name"":""gisuser"",""name"":""GISuser GeoTech News"",""id"":16958615,""id_str"":""16958615"",""indices"":[1,9]}]}","note","object:search.twitter.com,2005:494276921765543936","""@gisuser: Online Mapping for Beginners – The Map Academy from CartoDB: TweetSomething innovative from… http://t.co/Zhh882TV2F""","2014-07-30T00:22:56.000Z","http://twitter.com/httsan/statuses/494276921765543936","place","Haslett, MI","https://api.twitter.com/1.1/geo/id/4399b5004a3b4d9a.json","{""type"":""Polygon"",""coordinates"":[[[-84.447506,42.731229],[-84.447506,42.769688],[-84.363432,42.769688],[-84.363432,42.731229]]]}",,"Haslett","{""urls"":[{""url"":""http://t.co/Zhh882TV2F"",""expanded_url"":""http://blog.gisuser.com/2014/07/29/online-mapping-for-beginners-the-map-academy-from-cartodb/?utm_source=feedburner&utm_medium=twitter&utm_campaign=Feed:%20gisuser%20(GISUser.com%20-%20GIS,%20Mapping,%20Geospatial,%20and%20location%20technology%20news)"",""expanded_status"":200}],""klout_score"":51,""language"":{""value"":""en""}}","{""type"":""Point"",""coordinates"":[-84.4221123,42.7539688]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:494274906528288769","post","http://twitter.com/carygeo/statuses/494274906528288769","Attention mappers: iframe, #javascript, app development examples from the #CartoDB API/team! Well done @andrewxhill. http://t.co/cJM5fzgoyL","activity","2014-07-30T00:14:55.000Z","0","medium","en","0","person","id:twitter.com:2641688154","http://www.twitter.com/carygeo","Cary Greenwood","https://pbs.twimg.com/profile_images/488501325697544192/gFMxHuLV_normal.jpeg","geography, maps, data visualization, ui/ux design","2014-07-14T01:35:22.000Z","[{""href"":null,""rel"":""me""}]",,,"carygeo","[""en""]",,"109","46","0","47","false","Twitter for iPhone","http://twitter.com/download/iphone","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[34.13846767,-118.3620715]}","{""hashtags"":[{""text"":""javascript"",""indices"":[27,38]},{""text"":""CartoDB"",""indices"":[74,82]}],""symbols"":[],""urls"":[{""url"":""http://t.co/cJM5fzgoyL"",""expanded_url"":""http://vimeo.com/m/101936210"",""display_url"":""vimeo.com/m/101936210"",""indices"":[117,139]}],""user_mentions"":[{""screen_name"":""andrewxhill"",""name"":""Andrew W Hill"",""id"":19893224,""id_str"":""19893224"",""indices"":[103,115]}]}","note","object:search.twitter.com,2005:494274906528288769","Attention mappers: iframe, #javascript, app development examples from the #CartoDB API/team! Well done @andrewxhill. http://t.co/cJM5fzgoyL","2014-07-30T00:14:55.000Z","http://twitter.com/carygeo/statuses/494274906528288769","place","Los Angeles, CA","https://api.twitter.com/1.1/geo/id/3b77caf94bfc81fe.json","{""type"":""Polygon"",""coordinates"":[[[-118.668404,33.704538],[-118.668404,34.330724],[-118.155409,34.330724],[-118.155409,33.704538]]]}",,"Los Angeles","{""urls"":[{""url"":""http://t.co/cJM5fzgoyL"",""expanded_url"":""http://vimeo.com/101936210"",""expanded_status"":200}],""klout_score"":31,""language"":{""value"":""en""}}","{""type"":""Point"",""coordinates"":[-118.3620715,34.13846767]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:494194150133481472","post","http://twitter.com/juanjeojeda/statuses/494194150133481472","@dig_geo_com Your welcome :-) Keep in mind that some parts of @cartoDB still doesn't work fine. We're waiting for update from our provider.","activity","2014-07-29T18:54:01.000Z","0","medium","en","0","person","id:twitter.com:12679112","http://www.twitter.com/juanjeojeda","Juanje Ojeda ","https://pbs.twimg.com/profile_images/46025702/juanje_normal.jpg","Canarian guy who likes the open knowledge, the freedom and learn things every day. #OpenSource #DevOps #PaaS #Canarias","2008-01-25T11:49:15.000Z","[{""href"":""http://about.me/juanje"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Las Palmas de Gran Canaria""}","3600","juanjeojeda","[""es""]","London","416","1006","86","17700","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com","http://twitter.com/dig_geo_com/statuses/494191759078211584",,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""dig_geo_com"",""name"":""digital-geography"",""id"":231016902,""id_str"":""231016902"",""indices"":[0,12]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[62,70]}],""symbols"":[]}","note","object:search.twitter.com,2005:494194150133481472","@dig_geo_com Your welcome :-) Keep in mind that some parts of @cartoDB still doesn't work fine. We're waiting for update from our provider.","2014-07-29T18:54:01.000Z","http://twitter.com/juanjeojeda/statuses/494194150133481472","place","Las Palmas de Gran Canaria, Las Palmas","https://api.twitter.com/1.1/geo/id/00ab08eef1b62e92.json","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}",,"Las Palmas de Gran Canaria","{""klout_score"":56,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:494191583240388608","post","http://twitter.com/juanjeojeda/statuses/494191583240388608","@dig_geo_com Hi, @cartoDB blog is already back. We had some issues with our hosting provider. Sorry for the inconveniences.","activity","2014-07-29T18:43:49.000Z","0","medium","en","0","person","id:twitter.com:12679112","http://www.twitter.com/juanjeojeda","Juanje Ojeda ","https://pbs.twimg.com/profile_images/46025702/juanje_normal.jpg","Canarian guy who likes the open knowledge, the freedom and learn things every day. #OpenSource #DevOps #PaaS #Canarias","2008-01-25T11:49:15.000Z","[{""href"":""http://about.me/juanje"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Las Palmas de Gran Canaria""}","3600","juanjeojeda","[""es""]","London","416","1006","86","17698","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com","http://twitter.com/dig_geo_com/statuses/494181592685117442",,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""dig_geo_com"",""name"":""digital-geography"",""id"":231016902,""id_str"":""231016902"",""indices"":[0,12]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[17,25]}],""symbols"":[]}","note","object:search.twitter.com,2005:494191583240388608","@dig_geo_com Hi, @cartoDB blog is already back. We had some issues with our hosting provider. Sorry for the inconveniences.","2014-07-29T18:43:49.000Z","http://twitter.com/juanjeojeda/statuses/494191583240388608","place","Las Palmas de Gran Canaria, Las Palmas","https://api.twitter.com/1.1/geo/id/00ab08eef1b62e92.json","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}",,"Las Palmas de Gran Canaria","{""klout_score"":56,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:494141648922611712","post","http://twitter.com/spara/statuses/494141648922611712","@vtcraghead @briantimoney @billdollins CartoDB needs to cut a deal with a cloud provider, they’re a bit too parsimonious on disk space","activity","2014-07-29T15:25:24.000Z","0","medium","en","0","person","id:twitter.com:14629939","http://www.twitter.com/spara","spara","https://pbs.twimg.com/profile_images/469468370413166592/uuOHhLby_normal.jpeg","Ut mitterent eos in faciem.","2008-05-02T19:24:50.000Z","[{""href"":""http://sproke.blogspot.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""San Antonio Francisco""}","-25200","spara","[""en""]","Pacific Time (US & Canada)","1311","1276","128","24585","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com","http://twitter.com/vtcraghead/statuses/494140334226833408",,"{""hashtags"":[],""symbols"":[],""urls"":[],""user_mentions"":[{""screen_name"":""vtcraghead"",""name"":""Bill Morris"",""id"":278873782,""id_str"":""278873782"",""indices"":[0,11]},{""screen_name"":""briantimoney"",""name"":""Brian Timoney"",""id"":17546328,""id_str"":""17546328"",""indices"":[12,25]},{""screen_name"":""billdollins"",""name"":""Bill Dollins"",""id"":12405802,""id_str"":""12405802"",""indices"":[26,38]}]}","note","object:search.twitter.com,2005:494141648922611712","@vtcraghead @briantimoney @billdollins CartoDB needs to cut a deal with a cloud provider, they’re a bit too parsimonious on disk space","2014-07-29T15:25:24.000Z","http://twitter.com/spara/statuses/494141648922611712","place","Highland Park, San Antonio","https://api.twitter.com/1.1/geo/id/097c1754d9aa7b39.json","{""type"":""Polygon"",""coordinates"":[[[-98.478227,29.384112],[-98.478227,29.401578],[-98.448015,29.401578],[-98.448015,29.384112]]]}",,"Highland Park","{""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-98.478227,29.384112],[-98.478227,29.401578],[-98.448015,29.401578],[-98.448015,29.384112]]]}","Category 1","uno, @dos, #tres"
"tag:search.twitter.com,2005:496274441584668672","post","http://twitter.com/erictheise/statuses/496274441584668672","sad to be the guy that knocked ""locahost"" out of @cartoDB's docs for #Odyssey.js; i've certainly had that sentiment towards my dev machine.","activity","2014-08-04T12:40:21.000Z","0","medium","en","0","person","id:twitter.com:94817737","http://www.twitter.com/erictheise","Eric Theise","https://pbs.twimg.com/profile_images/1994194116/Theise_normal.jpg","Software engineer; photographer, experimental filmmaker, cartographer, geographer, vocalizer, yoga student, reader & writer, eater & drinker.","2009-12-05T16:06:12.000Z","[{""href"":null,""rel"":""me""}]","{""objectType"":""place"",""displayName"":""San Francisco""}","-25200","erictheise","[""en""]","Pacific Time (US & Canada)","698","306","25","1322","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[{""text"":""Odyssey"",""indices"":[69,77]}],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[49,57]}],""symbols"":[]}","note","object:search.twitter.com,2005:496274441584668672","sad to be the guy that knocked ""locahost"" out of @cartoDB's docs for #Odyssey.js; i've certainly had that sentiment towards my dev machine.","2014-08-04T12:40:21.000Z","http://twitter.com/erictheise/statuses/496274441584668672","place","San Francisco, CA","https://api.twitter.com/1.1/geo/id/5a110d312052166f.json","{""type"":""Polygon"",""coordinates"":[[[-122.514926,37.708075],[-122.514926,37.833238],[-122.328001,37.833238],[-122.328001,37.708075]]]}",,"San Francisco","{""klout_score"":41,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-122.514926,37.708075],[-122.514926,37.833238],[-122.328001,37.833238],[-122.328001,37.708075]]]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:495565956908064769","post","http://twitter.com/Jmholleran/statuses/495565956908064769","FrackMap #map http://t.co/BLWl3zYKTn","activity","2014-08-02T13:45:05.000Z","0","medium","en","0","person","id:twitter.com:27128139","http://www.twitter.com/Jmholleran","John Munro Holleran","https://pbs.twimg.com/profile_images/460662090819051520/lzLtBzUb_normal.png","Modest! intelligent, articulate & generally just amazing! Trade Unionist, learning facilitator 'n that","2009-03-27T23:36:39.000Z","[{""href"":null,""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Glasgow/Scotland""}",,"Jmholleran","[""en""]",,"411","192","1","1643","false","Twitter for Android","http://twitter.com/download/android","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[55.8641,-4.282299]}","{""hashtags"":[{""text"":""map"",""indices"":[9,13]}],""trends"":[],""urls"":[{""url"":""http://t.co/BLWl3zYKTn"",""expanded_url"":""http://energydesk.cartodb.com/viz/18162194-ecbd-11e3-aba1-0e230854a1cb/embed_map?title=true&description=true&search=false&shareable=true&cartodb_logo=true&layer_selector=false&legends=true&scrollwheel=true&fullscreen=true&sublayer_options=1%7C1%7C1%7C1&sql=&sw_lat=49.97948776108648&sw_lon=-12.7001953125&ne_lat=56.80087831233043&ne_lon=8.96484375"",""display_url"":""energydesk.cartodb.com/viz/18162194-e…"",""indices"":[14,36]}],""user_mentions"":[],""symbols"":[]}","note","object:search.twitter.com,2005:495565956908064769","FrackMap #map http://t.co/BLWl3zYKTn","2014-08-02T13:45:05.000Z","http://twitter.com/Jmholleran/statuses/495565956908064769","place","Glasgow","https://api.twitter.com/1.1/geo/id/791e00bcadc4615f.json","{""type"":""Polygon"",""coordinates"":[[[-4.3932845,55.796184],[-4.3932845,55.9204214],[-4.0902182,55.9204214],[-4.0902182,55.796184]]]}",,"Glasgow","{""urls"":[{""url"":""http://t.co/BLWl3zYKTn"",""expanded_url"":""http://energydesk.cartodb.com/viz/18162194-ecbd-11e3-aba1-0e230854a1cb/embed_map?title=true&description=true&search=false&shareable=true&cartodb_logo=true&layer_selector=false&legends=true&scrollwheel=true&fullscreen=true&sublayer_options=1%7C1%7C1%7C1&sql&sw_lat=49.97948776108648&sw_lon=-12.7001953125&ne_lat=56.80087831233043&ne_lon=8.96484375"",""expanded_status"":200}],""klout_score"":23}","{""type"":""Point"",""coordinates"":[-4.282299,55.8641]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:494521462682685440","post","http://twitter.com/SonOfJorEl/statuses/494521462682685440","@iriberri1 @cartoDB you are saying it is not supported, but you have an entire blog post about said feature : http://t.co/gvuUWv9sge","activity","2014-07-30T16:34:39.000Z","0","medium","en","0","person","id:twitter.com:15628310","http://www.twitter.com/SonOfJorEl","Joey Schluchter","https://pbs.twimg.com/profile_images/466697750013562880/ngwKpO1j_normal.jpeg","Recovering golfer. Helping feed the world to software. Still love persimmon.","2008-07-28T05:34:03.000Z","[{""href"":""http://www.exzeo.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Clearwater, FL""}","-14400","SonOfJorEl","[""en""]","Eastern Time (US & Canada)","368","2053","5","2143","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com","http://twitter.com/iriberri1/statuses/494488277765091328",,"{""hashtags"":[],""symbols"":[],""urls"":[{""url"":""http://t.co/gvuUWv9sge"",""expanded_url"":""http://blog.cartodb.com/got-files-weve-got-a-import-api/"",""display_url"":""blog.cartodb.com/got-files-weve…"",""indices"":[110,132]}],""user_mentions"":[{""screen_name"":""iriberri1"",""name"":""Carla"",""id"":102197411,""id_str"":""102197411"",""indices"":[0,10]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[11,19]}]}","note","object:search.twitter.com,2005:494521462682685440","@iriberri1 @cartoDB you are saying it is not supported, but you have an entire blog post about said feature : http://t.co/gvuUWv9sge","2014-07-30T16:34:39.000Z","http://twitter.com/SonOfJorEl/statuses/494521462682685440","place","Tampa, FL","https://api.twitter.com/1.1/geo/id/dc62519fda13b4ec.json","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}",,"Tampa","{""urls"":[{""url"":""http://t.co/gvuUWv9sge"",""expanded_url"":""http://blog.cartodb.com/got-files-weve-got-a-import-api/"",""expanded_status"":200}],""klout_score"":42,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:494511938563342336","post","http://twitter.com/xavijam/statuses/494511938563342336",". @Javier is giving birth last @cartodb feature. All I can say is … GIF over maps! http://t.co/oC3DEK5cP6","activity","2014-07-30T15:56:48.000Z","0","medium","en","0","person","id:twitter.com:26775253","http://www.twitter.com/xavijam","Javier Álvarez","https://pbs.twimg.com/profile_images/3025038956/a2919d353eb2b1a22756d7ac79847480_normal.jpeg","I love gentoo penguins","2009-03-26T15:36:52.000Z","[{""href"":""http://xavij.am"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Antarctic Peninsula""}","7200","xavijam","[""en""]","Madrid","250","395","26","4981","false","Tweetbot for iΟS","http://tapbots.com/tweetbot","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""javier"",""name"":""Javier Arce"",""id"":39083,""id_str"":""39083"",""indices"":[2,9]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[31,39]}],""symbols"":[],""media"":[{""id"":494511934629093378,""id_str"":""494511934629093378"",""indices"":[83,105],""media_url"":""http://pbs.twimg.com/media/Btzb_AjCIAI0cLr.jpg"",""media_url_https"":""https://pbs.twimg.com/media/Btzb_AjCIAI0cLr.jpg"",""url"":""http://t.co/oC3DEK5cP6"",""display_url"":""pic.twitter.com/oC3DEK5cP6"",""expanded_url"":""http://twitter.com/xavijam/status/494511938563342336/photo/1"",""type"":""photo"",""sizes"":{""small"":{""w"":340,""h"":255,""resize"":""fit""},""medium"":{""w"":600,""h"":450,""resize"":""fit""},""large"":{""w"":1024,""h"":768,""resize"":""fit""},""thumb"":{""w"":150,""h"":150,""resize"":""crop""}}}]}","note","object:search.twitter.com,2005:494511938563342336",". @Javier is giving birth last @cartodb feature. All I can say is … GIF over maps! http://t.co/oC3DEK5cP6","2014-07-30T15:56:48.000Z","http://twitter.com/xavijam/statuses/494511938563342336","place","Trafalgar","https://api.twitter.com/1.1/geo/id/0144b1172069289c.json","{""type"":""Polygon"",""coordinates"":[[[-3.70582,40.4275708],[-3.70582,40.4386747],[-3.6956326,40.4386747],[-3.6956326,40.4275708]]]}",,"Trafalgar","{""urls"":[{""url"":""http://t.co/oC3DEK5cP6"",""expanded_url"":""http://twitter.com/xavijam/status/494511938563342336/photo/1"",""expanded_status"":200}],""klout_score"":39,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-3.70582,40.4275708],[-3.70582,40.4386747],[-3.6956326,40.4386747],[-3.6956326,40.4275708]]]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:494485754848894978","post","http://twitter.com/SonOfJorEl/statuses/494485754848894978","@cartoDB I’ve emailed support (which you claim has a 24hr turnaround) with minimal response. I am a paying customer ($149 mo). For how long?","activity","2014-07-30T14:12:45.000Z","0","medium","en","0","person","id:twitter.com:15628310","http://www.twitter.com/SonOfJorEl","Joey Schluchter","https://pbs.twimg.com/profile_images/466697750013562880/ngwKpO1j_normal.jpeg","Recovering golfer. Helping feed the world to software. Still love persimmon.","2008-07-28T05:34:03.000Z","[{""href"":""http://www.exzeo.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Clearwater, FL""}","-14400","SonOfJorEl","[""en""]","Eastern Time (US & Canada)","368","2053","5","2142","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[0,8]}],""symbols"":[]}","note","object:search.twitter.com,2005:494485754848894978","@cartoDB I’ve emailed support (which you claim has a 24hr turnaround) with minimal response. I am a paying customer ($149 mo). For how long?","2014-07-30T14:12:45.000Z","http://twitter.com/SonOfJorEl/statuses/494485754848894978","place","Tampa, FL","https://api.twitter.com/1.1/geo/id/dc62519fda13b4ec.json","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}",,"Tampa","{""klout_score"":42,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:494276921765543936","post","http://twitter.com/httsan/statuses/494276921765543936","""@gisuser: Online Mapping for Beginners – The Map Academy from CartoDB: TweetSomething innovative from… http://t.co/Zhh882TV2F""","activity","2014-07-30T00:22:56.000Z","0","medium","en","0","person","id:twitter.com:28471178","http://www.twitter.com/httsan","hartanto","https://pbs.twimg.com/profile_images/2217327231/IMG0033_normal.jpg","Lahir di Indonesia | Nguli di @NEOnetBPPT @BPPTeknologi | Ngobrol di @RSGISForum @PadangZaitun | Nyantri di @MichiganStateU","2009-04-03T01:32:35.000Z","[{""href"":""http://about.me/httsan"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""East Lansing""}","-14400","httsan","[""en""]","Eastern Time (US & Canada)","976","873","12","25725","false","Twitter for Android","http://twitter.com/download/android","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[42.7539688,-84.4221123]}","{""hashtags"":[],""symbols"":[],""urls"":[{""url"":""http://t.co/Zhh882TV2F"",""expanded_url"":""http://goo.gl/fb/xVB5rr"",""display_url"":""goo.gl/fb/xVB5rr"",""indices"":[104,126]}],""user_mentions"":[{""screen_name"":""gisuser"",""name"":""GISuser GeoTech News"",""id"":16958615,""id_str"":""16958615"",""indices"":[1,9]}]}","note","object:search.twitter.com,2005:494276921765543936","""@gisuser: Online Mapping for Beginners – The Map Academy from CartoDB: TweetSomething innovative from… http://t.co/Zhh882TV2F""","2014-07-30T00:22:56.000Z","http://twitter.com/httsan/statuses/494276921765543936","place","Haslett, MI","https://api.twitter.com/1.1/geo/id/4399b5004a3b4d9a.json","{""type"":""Polygon"",""coordinates"":[[[-84.447506,42.731229],[-84.447506,42.769688],[-84.363432,42.769688],[-84.363432,42.731229]]]}",,"Haslett","{""urls"":[{""url"":""http://t.co/Zhh882TV2F"",""expanded_url"":""http://blog.gisuser.com/2014/07/29/online-mapping-for-beginners-the-map-academy-from-cartodb/?utm_source=feedburner&utm_medium=twitter&utm_campaign=Feed:%20gisuser%20(GISUser.com%20-%20GIS,%20Mapping,%20Geospatial,%20and%20location%20technology%20news)"",""expanded_status"":200}],""klout_score"":51,""language"":{""value"":""en""}}","{""type"":""Point"",""coordinates"":[-84.4221123,42.7539688]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:494274906528288769","post","http://twitter.com/carygeo/statuses/494274906528288769","Attention mappers: iframe, #javascript, app development examples from the #CartoDB API/team! Well done @andrewxhill. http://t.co/cJM5fzgoyL","activity","2014-07-30T00:14:55.000Z","0","medium","en","0","person","id:twitter.com:2641688154","http://www.twitter.com/carygeo","Cary Greenwood","https://pbs.twimg.com/profile_images/488501325697544192/gFMxHuLV_normal.jpeg","geography, maps, data visualization, ui/ux design","2014-07-14T01:35:22.000Z","[{""href"":null,""rel"":""me""}]",,,"carygeo","[""en""]",,"109","46","0","47","false","Twitter for iPhone","http://twitter.com/download/iphone","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[34.13846767,-118.3620715]}","{""hashtags"":[{""text"":""javascript"",""indices"":[27,38]},{""text"":""CartoDB"",""indices"":[74,82]}],""symbols"":[],""urls"":[{""url"":""http://t.co/cJM5fzgoyL"",""expanded_url"":""http://vimeo.com/m/101936210"",""display_url"":""vimeo.com/m/101936210"",""indices"":[117,139]}],""user_mentions"":[{""screen_name"":""andrewxhill"",""name"":""Andrew W Hill"",""id"":19893224,""id_str"":""19893224"",""indices"":[103,115]}]}","note","object:search.twitter.com,2005:494274906528288769","Attention mappers: iframe, #javascript, app development examples from the #CartoDB API/team! Well done @andrewxhill. http://t.co/cJM5fzgoyL","2014-07-30T00:14:55.000Z","http://twitter.com/carygeo/statuses/494274906528288769","place","Los Angeles, CA","https://api.twitter.com/1.1/geo/id/3b77caf94bfc81fe.json","{""type"":""Polygon"",""coordinates"":[[[-118.668404,33.704538],[-118.668404,34.330724],[-118.155409,34.330724],[-118.155409,33.704538]]]}",,"Los Angeles","{""urls"":[{""url"":""http://t.co/cJM5fzgoyL"",""expanded_url"":""http://vimeo.com/101936210"",""expanded_status"":200}],""klout_score"":31,""language"":{""value"":""en""}}","{""type"":""Point"",""coordinates"":[-118.3620715,34.13846767]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:494194150133481472","post","http://twitter.com/juanjeojeda/statuses/494194150133481472","@dig_geo_com Your welcome :-) Keep in mind that some parts of @cartoDB still doesn't work fine. We're waiting for update from our provider.","activity","2014-07-29T18:54:01.000Z","0","medium","en","0","person","id:twitter.com:12679112","http://www.twitter.com/juanjeojeda","Juanje Ojeda ","https://pbs.twimg.com/profile_images/46025702/juanje_normal.jpg","Canarian guy who likes the open knowledge, the freedom and learn things every day. #OpenSource #DevOps #PaaS #Canarias","2008-01-25T11:49:15.000Z","[{""href"":""http://about.me/juanje"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Las Palmas de Gran Canaria""}","3600","juanjeojeda","[""es""]","London","416","1006","86","17700","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com","http://twitter.com/dig_geo_com/statuses/494191759078211584",,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""dig_geo_com"",""name"":""digital-geography"",""id"":231016902,""id_str"":""231016902"",""indices"":[0,12]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[62,70]}],""symbols"":[]}","note","object:search.twitter.com,2005:494194150133481472","@dig_geo_com Your welcome :-) Keep in mind that some parts of @cartoDB still doesn't work fine. We're waiting for update from our provider.","2014-07-29T18:54:01.000Z","http://twitter.com/juanjeojeda/statuses/494194150133481472","place","Las Palmas de Gran Canaria, Las Palmas","https://api.twitter.com/1.1/geo/id/00ab08eef1b62e92.json","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}",,"Las Palmas de Gran Canaria","{""klout_score"":56,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:494191583240388608","post","http://twitter.com/juanjeojeda/statuses/494191583240388608","@dig_geo_com Hi, @cartoDB blog is already back. We had some issues with our hosting provider. Sorry for the inconveniences.","activity","2014-07-29T18:43:49.000Z","0","medium","en","0","person","id:twitter.com:12679112","http://www.twitter.com/juanjeojeda","Juanje Ojeda ","https://pbs.twimg.com/profile_images/46025702/juanje_normal.jpg","Canarian guy who likes the open knowledge, the freedom and learn things every day. #OpenSource #DevOps #PaaS #Canarias","2008-01-25T11:49:15.000Z","[{""href"":""http://about.me/juanje"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Las Palmas de Gran Canaria""}","3600","juanjeojeda","[""es""]","London","416","1006","86","17698","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com","http://twitter.com/dig_geo_com/statuses/494181592685117442",,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""dig_geo_com"",""name"":""digital-geography"",""id"":231016902,""id_str"":""231016902"",""indices"":[0,12]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[17,25]}],""symbols"":[]}","note","object:search.twitter.com,2005:494191583240388608","@dig_geo_com Hi, @cartoDB blog is already back. We had some issues with our hosting provider. Sorry for the inconveniences.","2014-07-29T18:43:49.000Z","http://twitter.com/juanjeojeda/statuses/494191583240388608","place","Las Palmas de Gran Canaria, Las Palmas","https://api.twitter.com/1.1/geo/id/00ab08eef1b62e92.json","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}",,"Las Palmas de Gran Canaria","{""klout_score"":56,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:494141648922611712","post","http://twitter.com/spara/statuses/494141648922611712","@vtcraghead @briantimoney @billdollins CartoDB needs to cut a deal with a cloud provider, they’re a bit too parsimonious on disk space","activity","2014-07-29T15:25:24.000Z","0","medium","en","0","person","id:twitter.com:14629939","http://www.twitter.com/spara","spara","https://pbs.twimg.com/profile_images/469468370413166592/uuOHhLby_normal.jpeg","Ut mitterent eos in faciem.","2008-05-02T19:24:50.000Z","[{""href"":""http://sproke.blogspot.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""San Antonio Francisco""}","-25200","spara","[""en""]","Pacific Time (US & Canada)","1311","1276","128","24585","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com","http://twitter.com/vtcraghead/statuses/494140334226833408",,"{""hashtags"":[],""symbols"":[],""urls"":[],""user_mentions"":[{""screen_name"":""vtcraghead"",""name"":""Bill Morris"",""id"":278873782,""id_str"":""278873782"",""indices"":[0,11]},{""screen_name"":""briantimoney"",""name"":""Brian Timoney"",""id"":17546328,""id_str"":""17546328"",""indices"":[12,25]},{""screen_name"":""billdollins"",""name"":""Bill Dollins"",""id"":12405802,""id_str"":""12405802"",""indices"":[26,38]}]}","note","object:search.twitter.com,2005:494141648922611712","@vtcraghead @briantimoney @billdollins CartoDB needs to cut a deal with a cloud provider, they’re a bit too parsimonious on disk space","2014-07-29T15:25:24.000Z","http://twitter.com/spara/statuses/494141648922611712","place","Highland Park, San Antonio","https://api.twitter.com/1.1/geo/id/097c1754d9aa7b39.json","{""type"":""Polygon"",""coordinates"":[[[-98.478227,29.384112],[-98.478227,29.401578],[-98.448015,29.401578],[-98.448015,29.384112]]]}",,"Highland Park","{""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-98.478227,29.384112],[-98.478227,29.401578],[-98.448015,29.401578],[-98.448015,29.384112]]]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:496274441584668672","post","http://twitter.com/erictheise/statuses/496274441584668672","sad to be the guy that knocked ""locahost"" out of @cartoDB's docs for #Odyssey.js; i've certainly had that sentiment towards my dev machine.","activity","2014-08-04T12:40:21.000Z","0","medium","en","0","person","id:twitter.com:94817737","http://www.twitter.com/erictheise","Eric Theise","https://pbs.twimg.com/profile_images/1994194116/Theise_normal.jpg","Software engineer; photographer, experimental filmmaker, cartographer, geographer, vocalizer, yoga student, reader & writer, eater & drinker.","2009-12-05T16:06:12.000Z","[{""href"":null,""rel"":""me""}]","{""objectType"":""place"",""displayName"":""San Francisco""}","-25200","erictheise","[""en""]","Pacific Time (US & Canada)","698","306","25","1322","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[{""text"":""Odyssey"",""indices"":[69,77]}],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[49,57]}],""symbols"":[]}","note","object:search.twitter.com,2005:496274441584668672","sad to be the guy that knocked ""locahost"" out of @cartoDB's docs for #Odyssey.js; i've certainly had that sentiment towards my dev machine.","2014-08-04T12:40:21.000Z","http://twitter.com/erictheise/statuses/496274441584668672","place","San Francisco, CA","https://api.twitter.com/1.1/geo/id/5a110d312052166f.json","{""type"":""Polygon"",""coordinates"":[[[-122.514926,37.708075],[-122.514926,37.833238],[-122.328001,37.833238],[-122.328001,37.708075]]]}",,"San Francisco","{""klout_score"":41,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-122.514926,37.708075],[-122.514926,37.833238],[-122.328001,37.833238],[-122.328001,37.708075]]]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:495565956908064769","post","http://twitter.com/Jmholleran/statuses/495565956908064769","FrackMap #map http://t.co/BLWl3zYKTn","activity","2014-08-02T13:45:05.000Z","0","medium","en","0","person","id:twitter.com:27128139","http://www.twitter.com/Jmholleran","John Munro Holleran","https://pbs.twimg.com/profile_images/460662090819051520/lzLtBzUb_normal.png","Modest! intelligent, articulate & generally just amazing! Trade Unionist, learning facilitator 'n that","2009-03-27T23:36:39.000Z","[{""href"":null,""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Glasgow/Scotland""}",,"Jmholleran","[""en""]",,"411","192","1","1643","false","Twitter for Android","http://twitter.com/download/android","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[55.8641,-4.282299]}","{""hashtags"":[{""text"":""map"",""indices"":[9,13]}],""trends"":[],""urls"":[{""url"":""http://t.co/BLWl3zYKTn"",""expanded_url"":""http://energydesk.cartodb.com/viz/18162194-ecbd-11e3-aba1-0e230854a1cb/embed_map?title=true&description=true&search=false&shareable=true&cartodb_logo=true&layer_selector=false&legends=true&scrollwheel=true&fullscreen=true&sublayer_options=1%7C1%7C1%7C1&sql=&sw_lat=49.97948776108648&sw_lon=-12.7001953125&ne_lat=56.80087831233043&ne_lon=8.96484375"",""display_url"":""energydesk.cartodb.com/viz/18162194-e…"",""indices"":[14,36]}],""user_mentions"":[],""symbols"":[]}","note","object:search.twitter.com,2005:495565956908064769","FrackMap #map http://t.co/BLWl3zYKTn","2014-08-02T13:45:05.000Z","http://twitter.com/Jmholleran/statuses/495565956908064769","place","Glasgow","https://api.twitter.com/1.1/geo/id/791e00bcadc4615f.json","{""type"":""Polygon"",""coordinates"":[[[-4.3932845,55.796184],[-4.3932845,55.9204214],[-4.0902182,55.9204214],[-4.0902182,55.796184]]]}",,"Glasgow","{""urls"":[{""url"":""http://t.co/BLWl3zYKTn"",""expanded_url"":""http://energydesk.cartodb.com/viz/18162194-ecbd-11e3-aba1-0e230854a1cb/embed_map?title=true&description=true&search=false&shareable=true&cartodb_logo=true&layer_selector=false&legends=true&scrollwheel=true&fullscreen=true&sublayer_options=1%7C1%7C1%7C1&sql&sw_lat=49.97948776108648&sw_lon=-12.7001953125&ne_lat=56.80087831233043&ne_lon=8.96484375"",""expanded_status"":200}],""klout_score"":23}","{""type"":""Point"",""coordinates"":[-4.282299,55.8641]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:494521462682685440","post","http://twitter.com/SonOfJorEl/statuses/494521462682685440","@iriberri1 @cartoDB you are saying it is not supported, but you have an entire blog post about said feature : http://t.co/gvuUWv9sge","activity","2014-07-30T16:34:39.000Z","0","medium","en","0","person","id:twitter.com:15628310","http://www.twitter.com/SonOfJorEl","Joey Schluchter","https://pbs.twimg.com/profile_images/466697750013562880/ngwKpO1j_normal.jpeg","Recovering golfer. Helping feed the world to software. Still love persimmon.","2008-07-28T05:34:03.000Z","[{""href"":""http://www.exzeo.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Clearwater, FL""}","-14400","SonOfJorEl","[""en""]","Eastern Time (US & Canada)","368","2053","5","2143","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com","http://twitter.com/iriberri1/statuses/494488277765091328",,"{""hashtags"":[],""symbols"":[],""urls"":[{""url"":""http://t.co/gvuUWv9sge"",""expanded_url"":""http://blog.cartodb.com/got-files-weve-got-a-import-api/"",""display_url"":""blog.cartodb.com/got-files-weve…"",""indices"":[110,132]}],""user_mentions"":[{""screen_name"":""iriberri1"",""name"":""Carla"",""id"":102197411,""id_str"":""102197411"",""indices"":[0,10]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[11,19]}]}","note","object:search.twitter.com,2005:494521462682685440","@iriberri1 @cartoDB you are saying it is not supported, but you have an entire blog post about said feature : http://t.co/gvuUWv9sge","2014-07-30T16:34:39.000Z","http://twitter.com/SonOfJorEl/statuses/494521462682685440","place","Tampa, FL","https://api.twitter.com/1.1/geo/id/dc62519fda13b4ec.json","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}",,"Tampa","{""urls"":[{""url"":""http://t.co/gvuUWv9sge"",""expanded_url"":""http://blog.cartodb.com/got-files-weve-got-a-import-api/"",""expanded_status"":200}],""klout_score"":42,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:494511938563342336","post","http://twitter.com/xavijam/statuses/494511938563342336",". @Javier is giving birth last @cartodb feature. All I can say is … GIF over maps! http://t.co/oC3DEK5cP6","activity","2014-07-30T15:56:48.000Z","0","medium","en","0","person","id:twitter.com:26775253","http://www.twitter.com/xavijam","Javier Álvarez","https://pbs.twimg.com/profile_images/3025038956/a2919d353eb2b1a22756d7ac79847480_normal.jpeg","I love gentoo penguins","2009-03-26T15:36:52.000Z","[{""href"":""http://xavij.am"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Antarctic Peninsula""}","7200","xavijam","[""en""]","Madrid","250","395","26","4981","false","Tweetbot for iΟS","http://tapbots.com/tweetbot","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""javier"",""name"":""Javier Arce"",""id"":39083,""id_str"":""39083"",""indices"":[2,9]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[31,39]}],""symbols"":[],""media"":[{""id"":494511934629093378,""id_str"":""494511934629093378"",""indices"":[83,105],""media_url"":""http://pbs.twimg.com/media/Btzb_AjCIAI0cLr.jpg"",""media_url_https"":""https://pbs.twimg.com/media/Btzb_AjCIAI0cLr.jpg"",""url"":""http://t.co/oC3DEK5cP6"",""display_url"":""pic.twitter.com/oC3DEK5cP6"",""expanded_url"":""http://twitter.com/xavijam/status/494511938563342336/photo/1"",""type"":""photo"",""sizes"":{""small"":{""w"":340,""h"":255,""resize"":""fit""},""medium"":{""w"":600,""h"":450,""resize"":""fit""},""large"":{""w"":1024,""h"":768,""resize"":""fit""},""thumb"":{""w"":150,""h"":150,""resize"":""crop""}}}]}","note","object:search.twitter.com,2005:494511938563342336",". @Javier is giving birth last @cartodb feature. All I can say is … GIF over maps! http://t.co/oC3DEK5cP6","2014-07-30T15:56:48.000Z","http://twitter.com/xavijam/statuses/494511938563342336","place","Trafalgar","https://api.twitter.com/1.1/geo/id/0144b1172069289c.json","{""type"":""Polygon"",""coordinates"":[[[-3.70582,40.4275708],[-3.70582,40.4386747],[-3.6956326,40.4386747],[-3.6956326,40.4275708]]]}",,"Trafalgar","{""urls"":[{""url"":""http://t.co/oC3DEK5cP6"",""expanded_url"":""http://twitter.com/xavijam/status/494511938563342336/photo/1"",""expanded_status"":200}],""klout_score"":39,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-3.70582,40.4275708],[-3.70582,40.4386747],[-3.6956326,40.4386747],[-3.6956326,40.4275708]]]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:494485754848894978","post","http://twitter.com/SonOfJorEl/statuses/494485754848894978","@cartoDB I’ve emailed support (which you claim has a 24hr turnaround) with minimal response. I am a paying customer ($149 mo). For how long?","activity","2014-07-30T14:12:45.000Z","0","medium","en","0","person","id:twitter.com:15628310","http://www.twitter.com/SonOfJorEl","Joey Schluchter","https://pbs.twimg.com/profile_images/466697750013562880/ngwKpO1j_normal.jpeg","Recovering golfer. Helping feed the world to software. Still love persimmon.","2008-07-28T05:34:03.000Z","[{""href"":""http://www.exzeo.com"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Clearwater, FL""}","-14400","SonOfJorEl","[""en""]","Eastern Time (US & Canada)","368","2053","5","2142","false","Tweetbot for Mac","http://tapbots.com/software/tweetbot/mac","service","Twitter","http://www.twitter.com",,,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[0,8]}],""symbols"":[]}","note","object:search.twitter.com,2005:494485754848894978","@cartoDB I’ve emailed support (which you claim has a 24hr turnaround) with minimal response. I am a paying customer ($149 mo). For how long?","2014-07-30T14:12:45.000Z","http://twitter.com/SonOfJorEl/statuses/494485754848894978","place","Tampa, FL","https://api.twitter.com/1.1/geo/id/dc62519fda13b4ec.json","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}",,"Tampa","{""klout_score"":42,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-82.648358,27.821353],[-82.648358,28.171245],[-82.289105,28.171245],[-82.289105,27.821353]]]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:494276921765543936","post","http://twitter.com/httsan/statuses/494276921765543936","""@gisuser: Online Mapping for Beginners – The Map Academy from CartoDB: TweetSomething innovative from… http://t.co/Zhh882TV2F""","activity","2014-07-30T00:22:56.000Z","0","medium","en","0","person","id:twitter.com:28471178","http://www.twitter.com/httsan","hartanto","https://pbs.twimg.com/profile_images/2217327231/IMG0033_normal.jpg","Lahir di Indonesia | Nguli di @NEOnetBPPT @BPPTeknologi | Ngobrol di @RSGISForum @PadangZaitun | Nyantri di @MichiganStateU","2009-04-03T01:32:35.000Z","[{""href"":""http://about.me/httsan"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""East Lansing""}","-14400","httsan","[""en""]","Eastern Time (US & Canada)","976","873","12","25725","false","Twitter for Android","http://twitter.com/download/android","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[42.7539688,-84.4221123]}","{""hashtags"":[],""symbols"":[],""urls"":[{""url"":""http://t.co/Zhh882TV2F"",""expanded_url"":""http://goo.gl/fb/xVB5rr"",""display_url"":""goo.gl/fb/xVB5rr"",""indices"":[104,126]}],""user_mentions"":[{""screen_name"":""gisuser"",""name"":""GISuser GeoTech News"",""id"":16958615,""id_str"":""16958615"",""indices"":[1,9]}]}","note","object:search.twitter.com,2005:494276921765543936","""@gisuser: Online Mapping for Beginners – The Map Academy from CartoDB: TweetSomething innovative from… http://t.co/Zhh882TV2F""","2014-07-30T00:22:56.000Z","http://twitter.com/httsan/statuses/494276921765543936","place","Haslett, MI","https://api.twitter.com/1.1/geo/id/4399b5004a3b4d9a.json","{""type"":""Polygon"",""coordinates"":[[[-84.447506,42.731229],[-84.447506,42.769688],[-84.363432,42.769688],[-84.363432,42.731229]]]}",,"Haslett","{""urls"":[{""url"":""http://t.co/Zhh882TV2F"",""expanded_url"":""http://blog.gisuser.com/2014/07/29/online-mapping-for-beginners-the-map-academy-from-cartodb/?utm_source=feedburner&utm_medium=twitter&utm_campaign=Feed:%20gisuser%20(GISUser.com%20-%20GIS,%20Mapping,%20Geospatial,%20and%20location%20technology%20news)"",""expanded_status"":200}],""klout_score"":51,""language"":{""value"":""en""}}","{""type"":""Point"",""coordinates"":[-84.4221123,42.7539688]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:494274906528288769","post","http://twitter.com/carygeo/statuses/494274906528288769","Attention mappers: iframe, #javascript, app development examples from the #CartoDB API/team! Well done @andrewxhill. http://t.co/cJM5fzgoyL","activity","2014-07-30T00:14:55.000Z","0","medium","en","0","person","id:twitter.com:2641688154","http://www.twitter.com/carygeo","Cary Greenwood","https://pbs.twimg.com/profile_images/488501325697544192/gFMxHuLV_normal.jpeg","geography, maps, data visualization, ui/ux design","2014-07-14T01:35:22.000Z","[{""href"":null,""rel"":""me""}]",,,"carygeo","[""en""]",,"109","46","0","47","false","Twitter for iPhone","http://twitter.com/download/iphone","service","Twitter","http://www.twitter.com",,"{""type"":""Point"",""coordinates"":[34.13846767,-118.3620715]}","{""hashtags"":[{""text"":""javascript"",""indices"":[27,38]},{""text"":""CartoDB"",""indices"":[74,82]}],""symbols"":[],""urls"":[{""url"":""http://t.co/cJM5fzgoyL"",""expanded_url"":""http://vimeo.com/m/101936210"",""display_url"":""vimeo.com/m/101936210"",""indices"":[117,139]}],""user_mentions"":[{""screen_name"":""andrewxhill"",""name"":""Andrew W Hill"",""id"":19893224,""id_str"":""19893224"",""indices"":[103,115]}]}","note","object:search.twitter.com,2005:494274906528288769","Attention mappers: iframe, #javascript, app development examples from the #CartoDB API/team! Well done @andrewxhill. http://t.co/cJM5fzgoyL","2014-07-30T00:14:55.000Z","http://twitter.com/carygeo/statuses/494274906528288769","place","Los Angeles, CA","https://api.twitter.com/1.1/geo/id/3b77caf94bfc81fe.json","{""type"":""Polygon"",""coordinates"":[[[-118.668404,33.704538],[-118.668404,34.330724],[-118.155409,34.330724],[-118.155409,33.704538]]]}",,"Los Angeles","{""urls"":[{""url"":""http://t.co/cJM5fzgoyL"",""expanded_url"":""http://vimeo.com/101936210"",""expanded_status"":200}],""klout_score"":31,""language"":{""value"":""en""}}","{""type"":""Point"",""coordinates"":[-118.3620715,34.13846767]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:494194150133481472","post","http://twitter.com/juanjeojeda/statuses/494194150133481472","@dig_geo_com Your welcome :-) Keep in mind that some parts of @cartoDB still doesn't work fine. We're waiting for update from our provider.","activity","2014-07-29T18:54:01.000Z","0","medium","en","0","person","id:twitter.com:12679112","http://www.twitter.com/juanjeojeda","Juanje Ojeda ","https://pbs.twimg.com/profile_images/46025702/juanje_normal.jpg","Canarian guy who likes the open knowledge, the freedom and learn things every day. #OpenSource #DevOps #PaaS #Canarias","2008-01-25T11:49:15.000Z","[{""href"":""http://about.me/juanje"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Las Palmas de Gran Canaria""}","3600","juanjeojeda","[""es""]","London","416","1006","86","17700","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com","http://twitter.com/dig_geo_com/statuses/494191759078211584",,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""dig_geo_com"",""name"":""digital-geography"",""id"":231016902,""id_str"":""231016902"",""indices"":[0,12]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[62,70]}],""symbols"":[]}","note","object:search.twitter.com,2005:494194150133481472","@dig_geo_com Your welcome :-) Keep in mind that some parts of @cartoDB still doesn't work fine. We're waiting for update from our provider.","2014-07-29T18:54:01.000Z","http://twitter.com/juanjeojeda/statuses/494194150133481472","place","Las Palmas de Gran Canaria, Las Palmas","https://api.twitter.com/1.1/geo/id/00ab08eef1b62e92.json","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}",,"Las Palmas de Gran Canaria","{""klout_score"":56,""language"":{""value"":""en""}}","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}","Category 2","aaa, bbb"
"tag:search.twitter.com,2005:494191583240388608","post","http://twitter.com/juanjeojeda/statuses/494191583240388608","@dig_geo_com Hi, @cartoDB blog is already back. We had some issues with our hosting provider. Sorry for the inconveniences.","activity","2014-07-29T18:43:49.000Z","0","medium","en","0","person","id:twitter.com:12679112","http://www.twitter.com/juanjeojeda","Juanje Ojeda ","https://pbs.twimg.com/profile_images/46025702/juanje_normal.jpg","Canarian guy who likes the open knowledge, the freedom and learn things every day. #OpenSource #DevOps #PaaS #Canarias","2008-01-25T11:49:15.000Z","[{""href"":""http://about.me/juanje"",""rel"":""me""}]","{""objectType"":""place"",""displayName"":""Las Palmas de Gran Canaria""}","3600","juanjeojeda","[""es""]","London","416","1006","86","17698","false","Twitter Web Client","http://twitter.com","service","Twitter","http://www.twitter.com","http://twitter.com/dig_geo_com/statuses/494181592685117442",,"{""hashtags"":[],""trends"":[],""urls"":[],""user_mentions"":[{""screen_name"":""dig_geo_com"",""name"":""digital-geography"",""id"":231016902,""id_str"":""231016902"",""indices"":[0,12]},{""screen_name"":""cartoDB"",""name"":""CartoDB"",""id"":241079136,""id_str"":""241079136"",""indices"":[17,25]}],""symbols"":[]}","note","object:search.twitter.com,2005:494191583240388608","@dig_geo_com Hi, @cartoDB blog is already back. We had some issues with our hosting provider. Sorry for the inconveniences.","2014-07-29T18:43:49.000Z","http://twitter.com/juanjeojeda/statuses/494191583240388608","place","Las Palmas de Gran Canaria, Las Palmas","https://api.twitter.com/1.1/geo/id/00ab08eef1b62e92.json","{""type"":""Polygon"",""coordinates"":[[[-15.5255036,28.0248125],[-15.5255036,28.1812125],[-15.3941943,28.1812125],[-15.3941943,28.0248125]]]}
Can't render this file because it contains an unexpected character in line 40 and column 1807.
@@ -0,0 +1,51 @@
require_relative '../../lib/datasources'
require_relative '../doubles/json_to_csv_converter'
include CartoDB::Datasources
describe CSVFileDumper do
before(:each) do
end
describe '#dumping data' do
it 'tests streaming dump' do
output_stream_name ='/tmp/csv_file_dumper_test.csv'
File.unlink(output_stream_name) if File.exists?(output_stream_name)
converter_mock = CartoDB::TwitterSearch::Doubles::JSONToCSVConverter.new
dumper = CSVFileDumper.new(converter_mock, false)
dumper.buffer_size=2
input_filename_1 = 'stream_input_1'
input_filename_2 = 'stream_input_2'
input_data1 = File.read(File.join(File.dirname(__FILE__), "../fixtures/#{input_filename_1}"))
input_data2 = File.read(File.join(File.dirname(__FILE__), "../fixtures/#{input_filename_2}"))
names_list = [ input_filename_1, input_filename_2 ]
stream = File.open(output_stream_name, 'wb')
dumper.begin_dump(input_filename_1)
dumper.dump(input_filename_1, [input_data1])
dumper.end_dump(input_filename_1)
dumper.begin_dump(input_filename_2)
dumper.dump(input_filename_2, [input_data2])
dumper.end_dump(input_filename_2)
dumper.merge_dumps_into_stream(names_list, stream)
stream.close
data = File.read(output_stream_name)
data.should eq "\n#{input_data1}\n#{input_data2}\n"
File.unlink(output_stream_name)
end
end
end
@@ -0,0 +1,103 @@
require_relative '../../lib/datasources'
require_relative '../doubles/organization'
require_relative '../doubles/user'
require_relative '../doubles/search_tweet'
require_relative '../doubles/data_import'
require_relative '../../../../lib/cartodb/logger'
include CartoDB::Datasources
describe Search::Twitter do
def get_config
{
'auth_required' => false,
'username' => '',
'password' => '',
'search_url' => 'http://fakeurl.carto',
}
end
before(:each) do
Typhoeus::Expectation.clear
end
describe '#search' do
it 'tests basic full search flow with streaming' do
user_quota = 100
user_mock = CartoDB::Datasources::Doubles::User.new({twitter_datasource_quota: user_quota})
data_import_mock = CartoDB::Datasources::Doubles::DataImport.new(id: '123456789', service_item_id: '987654321')
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
input_terms = terms_fixture
input_dates = dates_fixture
Typhoeus.stub(/fakeurl\.carto/) do |request|
accept = (request.options[:headers]||{})['Accept'] || 'application/json'
format = accept.split(',').first
request.base_url.should eq 'http://fakeurl.carto'
request.options[:params].key?(:pusblisher).should eq false
body = data_from_file('sample_tweets_v2.json')
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => format },
body: body
)
end
twitter_datasource.send :audit_entry, CartoDB::Datasources::Doubles::SearchTweet
twitter_datasource.data_import_item = data_import_mock
stream_location = '/tmp/sample_tweets_v2.csv'
File.unlink(stream_location) if File.exists?(stream_location)
stream = File.open(stream_location, 'wb')
twitter_datasource.stream_resource(::JSON.dump(
{
categories: input_terms[:categories],
dates: input_dates[:dates]
}
), stream)
stream.close
stored_data = data_from_file(stream_location, true)
stored_data.should eq data_from_file('sample_tweets_v2.csv')
File.unlink(stream_location)
end
end
protected
def terms_fixture
{
categories: [
{
category: '1',
terms: ['carto']
}
]
}
end
def dates_fixture
{
dates: {
fromDate: '2017-02-21',
fromHour: '11',
fromMin: '45',
toDate: '2017-02-21',
toHour: '12',
toMin: '00'
}
}
end
def data_from_file(filename, fullpath=false)
if fullpath
File.read(filename)
else
File.read(File.join(File.dirname(__FILE__), "../fixtures/#{filename}"))
end
end
end
@@ -0,0 +1,521 @@
require 'active_support/core_ext'
require_relative '../../lib/datasources'
require_relative '../doubles/user'
include CartoDB::Datasources
describe Url::ArcGIS do
before(:all) do
@url = 'http://myserver/arcgis/rest/services/MyFakeService/featurename'
@invalid_url = 'http://myserver/mysite/rest/myfakefolder/MyFakeService/featurename'
@user = CartoDB::Datasources::Doubles::User.new
end
before(:each) do
Typhoeus::Expectation.clear
end
describe '#set_data_from' do
it 'tests preparing the correct url from the one given from the UI' do
invalid_1 = 'http://myserver/services/MyFakeService/featurename/MapServer'
invalid_2 = 'myserver/services/MyFakeService/featurename/MapServer'
test1 = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer'
test2 = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer/'
test3 = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer/0'
test4 = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer/0?'
test5 = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer/0?q=blablabla'
test6 = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer?q=blablabla'
valid_map = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer'
valid_map_trailing_slash = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer/'
valid_layer = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/MapServer/0'
# Should be treated as ok
test7 = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/2314/'
valid_7 = 'http://myserver/arcgis/rest/services/MyFakeService/featurename/2314/'
arcgis = Url::ArcGIS.get_new(@user)
expect {
arcgis.send(:sanitize_id, invalid_1)
}.to raise_error InvalidInputDataError
expect {
arcgis.send(:sanitize_id, invalid_2)
}.to raise_error InvalidInputDataError
arcgis.send(:sanitize_id, test1).should eq valid_map
arcgis.send(:sanitize_id, test2).should eq valid_map_trailing_slash
arcgis.send(:sanitize_id, test3).should eq valid_layer
arcgis.send(:sanitize_id, test4).should eq valid_layer
arcgis.send(:sanitize_id, test5).should eq valid_layer
arcgis.send(:sanitize_id, test6).should eq valid_map
arcgis.send(:sanitize_id, test7).should eq valid_7
end
end
describe '#get_resource_metadata' do
it 'tests error scenarios' do
arcgis = Url::ArcGIS.get_new(@user)
sub_id = '0'
# 'general http error (non-200)'
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/layers/) do
Typhoeus::Response.new(
code: 400,
headers: { 'Content-Type' => 'application/json' },
body: ''
)
end
expect {
arcgis.get_resource_metadata(@url)
}.to raise_error DataDownloadError
# Stub layers request (so now works)
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/layers/) do |request|
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_layers.json"))
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => 'application/json' },
body: ::JSON.dump(::JSON.parse(body))
)
end
layers_data = arcgis.get_resource_metadata(@url)
layers_data_expected = {
id: @url,
subresources: [{
id: "#{@url}/#{sub_id}",
title: 'first layer'
}]
}
layers_data.should eq layers_data_expected
# 'fields' part
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/0/) do |request|
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_metadata_minimal.json"))
body = ::JSON.parse(body)
body.delete('fields')
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => 'application/json' },
body: ::JSON.dump(body)
)
end
expect {
arcgis.send(:get_subresource_metadata, @url, sub_id)
}.to raise_error ResponseError
# Another required field
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/0/) do |request|
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_metadata_minimal.json"))
body = ::JSON.parse(body)
body.delete('name')
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => 'application/json' },
body: ::JSON.dump(body)
)
end
expect {
arcgis.send(:get_subresource_metadata, @url, sub_id)
}.to raise_error ResponseError
# Invalid ArcGIS version
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/0/) do |request|
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_metadata_minimal.json"))
body = ::JSON.parse(body)
body['currentVersion'] = 9.0
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => 'application/json' },
body: ::JSON.dump(body)
)
end
expect {
arcgis.send(:get_subresource_metadata, @url, sub_id)
}.to raise_error InvalidServiceError
# Invalid ArcGIS URL
expect {
arcgis.send(:get_resource_metadata, @invalid_url)
}.to raise_error InvalidInputDataError
end
it 'tests metadata retrieval' do
arcgis = Url::ArcGIS.get_new(@user)
# Stub layers request (so now works)
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/layers/) do |request|
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_layers.json"))
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => 'application/json' },
body: ::JSON.dump(::JSON.parse(body))
)
end
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/0/) do |request|
accept = (request.options[:headers]||{})['Accept'] || 'application/json'
format = accept.split(',').first
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_metadata_minimal.json"))
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => format },
body: body
)
end
expected_metadata = {
:arcgis_version=>10.22,
:name=>"Test Feature",
:description=>"Sample metadata payload",
:type=>"Feature Layer",
:geometry_type=>"esriGeometryPolygon",
:copyright=>"CartoDB",
:fields=>[
{
:name => "OBJECTID",
:type => "esriFieldTypeOID"
},
{
:name => "NAME",
:type => "esriFieldTypeString"
}
],
:max_records_per_query=>1000,
:supported_formats=>["JSON", "AMF"],
:advanced_queries_supported=>true
}
expected_metadata_response = {
id: @url + '/0',
title: 'Test Feature',
url: nil,
service: Url::ArcGIS::DATASOURCE_NAME,
checksum: nil,
size: 0,
filename: 'test_feature.json'
}
# Multi-resource scenario already tested above
response = arcgis.get_resource_metadata(@url + '/0')
response.nil?.should be false
arcgis.metadata.should eq expected_metadata
response.should eq expected_metadata_response
end
end
describe '#get_resource' do
it 'tests the get_ids_list() private method with error scenarios' do
arcgis = Url::ArcGIS.get_new(@user)
id = arcgis.send(:sanitize_id, @url)
# 'general http error (non-200)'
Typhoeus.stub(/\/arcgis\/rest\//) do
Typhoeus::Response.new(
code: 400,
headers: { 'Content-Type' => 'application/json' },
body: ''
)
end
expect {
arcgis.send(:get_ids_list, id)
}.to raise_error DataDownloadError
# 'objectIds' not present
Typhoeus::Expectation.clear
Typhoeus.stub(/\/arcgis\/rest\//) do
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_ids_list.json"))
body = ::JSON.parse(body)
body.delete('objectIds')
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => 'application/json' },
body: ::JSON.dump(body)
)
end
expect {
arcgis.send(:get_ids_list, id)
}.to raise_error ResponseError
# 'objectIds' empty
Typhoeus::Expectation.clear
Typhoeus.stub(/\/arcgis\/rest\//) do
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_ids_list.json"))
body = ::JSON.parse(body)
body['objectIds'] = []
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => 'application/json' },
body: ::JSON.dump(body)
)
end
expect {
arcgis.send(:get_ids_list, id)
}.to raise_error ResponseError
end
it 'tests the get_ids_list() private method' do
arcgis = Url::ArcGIS.get_new(@user)
id = arcgis.send(:sanitize_id, @url)
Typhoeus.stub(/\/arcgis\/rest\//) do |request|
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_ids_list.json"))
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => 'application/json' },
body: body
)
end
expected_ids = [1,2,3,4,5,6,7,8,9,10]
respose_ids = arcgis.send(:get_ids_list, id)
respose_ids.nil?.should be false
respose_ids.should eq expected_ids
end
it 'tests the get_ids_list() private method on out-of-order ids' do
arcgis = Url::ArcGIS.get_new(@user)
id = arcgis.send(:sanitize_id, @url)
Typhoeus.stub(/\/arcgis\/rest\//) do |request|
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_unordered_ids_list.json"))
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => 'application/json' },
body: body
)
end
expected_ids = [1,2,3,4,5,6,7,8,9,10]
respose_ids = arcgis.send(:get_ids_list, id)
respose_ids.nil?.should be false
respose_ids.should eq expected_ids
end
it 'tests the get_by_ids() private method with error scenarios' do
arcgis = Url::ArcGIS.get_new(@user)
id = arcgis.send(:sanitize_id, @url)
# Empty ids
expect {
arcgis.send(:get_by_ids, id, [], [{ key: 'value' }])
}.to raise_error InvalidInputDataError
# Empty fields
expect {
arcgis.send(:get_by_ids, id, [1], [])
}.to raise_error InvalidInputDataError
# 'general http error (non-200)'
Typhoeus.stub(/\/arcgis\/rest\//) do
Typhoeus::Response.new(
code: 400,
headers: { 'Content-Type' => 'application/json' },
body: ''
)
end
expect {
arcgis.send(:get_by_ids, id, [1], [{ key: 'value' }])
}.to raise_error DataDownloadError
end
it 'tests the get_by_ids() private method' do
arcgis = Url::ArcGIS.get_new(@user)
id = arcgis.send(:sanitize_id, @url)
Typhoeus.stub(/\/arcgis\/rest\//) do
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_data_01.json"))
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => 'application/json' },
body: body
)
end
expected_response_data = {
geometryType: "esriGeometryPolygon",
spatialReference: {
"wkid" => 4326,
"latestWkid" => 4326
},
fields: [
{
"name" => "OBJECTID",
"type" => "esriFieldTypeOID",
"alias" => "OBJECTID"
},
{
"name" => "WDPAID",
"type" => "esriFieldTypeInteger",
"alias" => "WDPAID"
},
{
"name" => "NAME",
"type" => "esriFieldTypeString",
"alias" => "NAME",
"length" => 254
}
],
features: [
{"attributes"=>{"OBJECTID"=>1, "NAME"=>"Name of object 1"}, "geometry"=>{"fake"=>"geom"}},
{"attributes"=>{"OBJECTID"=>2, "NAME"=>"Name of object 2"}, "geometry"=>{"fake"=>"geom"}},
{"attributes"=>{"OBJECTID"=>3, "NAME"=>"Name of object 3"}, "geometry"=>{"fake"=>"geom"}}
]
}
ids_to_retrieve = [1,2,3]
# WDPAID also present, but left on purpose untouched
fields_to_retrieve = [{
name: 'OBJECTID',
type: 'esriFieldTypeOID'
},
{
name: 'NAME',
type: 'esriFieldTypeString'
}]
response_data = arcgis.send(:get_by_ids, id, ids_to_retrieve, fields_to_retrieve)
response_data.nil?.should eq false
response_data[:features].length.should eq 3
response_data.should eq expected_response_data
end
it 'tests retrieval of data' do
arcgis = Url::ArcGIS.get_new(@user)
feature_names = []
# Layers request
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/layers/) do |request|
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_layers.json"))
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => 'application/json' },
body: ::JSON.dump(::JSON.parse(body))
)
end
# Metadata of a layer
Typhoeus.stub(/\/arcgis\/rest\/services\/MyFakeService\/featurename\/0\?f=json/) do
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_metadata_minimal.json"))
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => 'application/json' },
body: body
)
end
# IDs list of a layer
Typhoeus.stub(/\/arcgis\/rest\/(.*)query\?where=/) do
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_ids_list_01.json"))
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => 'application/json' },
body: body
)
end
Typhoeus.stub(/\/arcgis\/rest\/(.*)query$/) do |response|
if response.options[:body][:objectIds].to_i == 1
# First item fetch of a layer
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_data_01.json"))
body = ::JSON.parse(body)
feature_names.push body['features'][0]['attributes']['NAME']
body['features'] = [ body['features'][0] ]
else
# Remaining items fetch of a layer, will not use :objectIds
body = File.read(File.join(File.dirname(__FILE__), "../fixtures/arcgis_data_01.json"))
body = ::JSON.parse(body)
feature_names.push body['features'][1]['attributes']['NAME']
feature_names.push body['features'][2]['attributes']['NAME']
body['features'] = [ body['features'][1], body['features'][2] ]
end
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => 'application/json' },
body: ::JSON.dump(body)
)
end
# 1) Retrieve lists of layers
metadata = arcgis.get_resource_metadata(@url)
# 2) Retrieve metadata of a specific layer (doesn't adds much, but to replicate flow)
item_metadata = arcgis.get_resource_metadata(metadata[:subresources].first[:id])
item_metadata.nil?.should eq false
# No more checks as will fail later if missing something
spatial_ref_expectation = {"wkid"=>4326, "latestWkid"=>4326}
initial_stream_data = arcgis.initial_stream(item_metadata[:id])
initial_stream_data.nil?.should eq false
initial_stream_data = ::JSON.parse(initial_stream_data)
initial_stream_data['geometryType'].should eq 'esriGeometryPolygon'
initial_stream_data['spatialReference'].should eq spatial_ref_expectation
initial_stream_data['fields'].count.should eq 3
initial_stream_data['fields'][0]['name'].should eq 'OBJECTID'
initial_stream_data['fields'][1]['name'].should eq 'WDPAID'
initial_stream_data['fields'][2]['name'].should eq 'NAME'
initial_stream_data['features'].count.should eq 1
initial_stream_data['features'][0]['attributes'].nil?.should eq false
initial_stream_data['features'][0]['attributes']['NAME'].should eq feature_names[0]
initial_stream_data['features'][0]['geometry'].nil?.should eq false
streamed_data = arcgis.stream_resource(item_metadata[:id])
streamed_data.nil?.should eq false
streamed_data = ::JSON.parse(streamed_data)
streamed_data['geometryType'].should eq 'esriGeometryPolygon'
streamed_data['spatialReference'].should eq spatial_ref_expectation
streamed_data['fields'].count.should eq 3
streamed_data['fields'][0]['name'].should eq 'OBJECTID'
streamed_data['fields'][1]['name'].should eq 'WDPAID'
streamed_data['fields'][2]['name'].should eq 'NAME'
streamed_data['features'].count.should eq 2
streamed_data['features'][0]['attributes']['NAME'].should eq feature_names[1]
streamed_data['features'][1]['attributes']['NAME'].should eq feature_names[2]
end
end
end
@@ -0,0 +1,31 @@
require_relative '../../lib/datasources'
require_relative '../doubles/user'
include CartoDB::Datasources
describe Url::Box do
def get_config
{
'box_host' => '',
'application_name' => '',
'client_id' => '',
'client_secret' => '',
'callback_url' => ''
}
end
describe '#filters' do
it 'test that filter sets correctly' do
user_mock = CartoDB::Datasources::Doubles::User.new
box_provider = Url::Box.get_new(get_config, user_mock)
box_provider.filter.should eq nil
# Filter to 'documents'
formats = ['csv', 'xls']
box_provider.filter = formats
box_provider.filter.should eq formats
end
end
end
@@ -0,0 +1,47 @@
require_relative '../../lib/datasources'
require_relative '../doubles/user'
include CartoDB::Datasources
describe Url::Dropbox do
def get_config
{
'app_key' => '',
'app_secret' => '',
'callback_url' => ''
}
end #get_config
describe '#filters' do
it 'test that filter options work correctly' do
user_mock = CartoDB::Datasources::Doubles::User.new
dropbox_provider = Url::Dropbox.get_new(get_config, user_mock)
# No filter = all formats allowed
filter = []
Url::Dropbox::FORMATS_TO_SEARCH_QUERIES.each do |id, search_queries|
search_queries.each do |search_query|
filter = filter.push(search_query)
end
end
dropbox_provider.filter.should eq filter
# Filter to 'documents'
filter = []
format_ids = [ Url::Dropbox::FORMAT_CSV, Url::Dropbox::FORMAT_EXCEL ]
Url::Dropbox::FORMATS_TO_SEARCH_QUERIES.each do |id, search_queries|
if format_ids.include?(id)
search_queries.each do |search_query|
filter = filter.push(search_query)
end
end
end
dropbox_provider.filter = format_ids
dropbox_provider.filter.should eq filter
end
end #run
end
@@ -0,0 +1,53 @@
require_relative '../../../../spec/rspec_configuration'
require_relative '../../lib/datasources'
require_relative '../doubles/user'
include CartoDB::Datasources
describe Url::GDrive do
def get_config
{
'application_name' => '',
'client_id' => '',
'client_secret' => '',
'callback_url' => 'http://localhost/callback'
}
end #get_config
describe '#filters' do
it 'test that filter options work correctly' do
# Stubs Google Drive client for connectionless testing
Google::Apis::DriveV2::DriveService.any_instance.stubs(:get_file)
Google::Apis::DriveV2::DriveService.any_instance.stubs(:export_file)
Google::Apis::DriveV2::DriveService.any_instance.stubs(:list_files)
user_mock = CartoDB::Datasources::Doubles::User.new
gdrive_provider = Url::GDrive.get_new(get_config, user_mock)
# No filter = all formats allowed
filter = []
Url::GDrive::FORMATS_TO_MIME_TYPES.each do |id, mime_types|
mime_types.each do |mime_type|
filter = filter.push(mime_type)
end
end
gdrive_provider.filter.should eq filter
# Filter to 'documents'
filter = []
format_ids = [ Url::GDrive::FORMAT_CSV, Url::GDrive::FORMAT_EXCEL ]
Url::GDrive::FORMATS_TO_MIME_TYPES.each do |id, mime_types|
if format_ids.include?(id)
mime_types.each do |mime_type|
filter = filter.push(mime_type)
end
end
end
gdrive_provider.filter = format_ids
gdrive_provider.filter.should eq filter
end
end #run
end
@@ -0,0 +1,372 @@
require_relative '../../lib/datasources'
require_relative '../doubles/organization'
require_relative '../doubles/user'
require_relative '../doubles/search_tweet'
require_relative '../doubles/data_import'
require_relative '../../../../lib/cartodb/logger'
include CartoDB::Datasources
describe Search::Twitter do
def get_config
{
'auth_required' => false,
'username' => '',
'password' => '',
'search_url' => 'http://fakeurl.carto',
}
end #get_config
before(:each) do
Typhoeus::Expectation.clear
end
describe '#filters' do
it 'tests max and total results filters' do
big_quota = 123456
user = CartoDB::Datasources::Doubles::User.new({
twitter_datasource_quota: big_quota
})
twitter_datasource = Search::Twitter.get_new(get_config, user)
maxresults_filter = twitter_datasource.send :build_maxresults_field, user
maxresults_filter.should eq CartoDB::TwitterSearch::SearchAPI::MAX_PAGE_RESULTS
totalresults_filter = twitter_datasource.send :build_total_results_field, user
totalresults_filter.should eq big_quota
small_quota = 13
user = CartoDB::Datasources::Doubles::User.new({
twitter_datasource_quota: small_quota,
soft_twitter_datasource_limit: false
})
twitter_datasource = Search::Twitter.get_new(get_config, user)
maxresults_filter = twitter_datasource.send :build_maxresults_field, user
maxresults_filter.should eq small_quota
totalresults_filter = twitter_datasource.send :build_total_results_field, user
totalresults_filter.should eq small_quota
user = CartoDB::Datasources::Doubles::User.new({
twitter_datasource_quota: small_quota,
soft_twitter_datasource_limit: true
})
maxresults_filter = twitter_datasource.send :build_maxresults_field, user
maxresults_filter.should eq CartoDB::TwitterSearch::SearchAPI::MAX_PAGE_RESULTS
totalresults_filter = twitter_datasource.send :build_total_results_field, user
totalresults_filter.should eq Search::Twitter::NO_TOTAL_RESULTS
end
it 'tests category filters' do
user_mock = CartoDB::Datasources::Doubles::User.new
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
input_terms = terms_fixture
expected_output_terms = [
{
Search::Twitter::CATEGORY_NAME_KEY => 'Category 1',
Search::Twitter::CATEGORY_TERMS_KEY => '(uno OR @dos OR #tres) (has:geo OR has:profile_geo)'
},
{
Search::Twitter::CATEGORY_NAME_KEY => 'Category 2',
Search::Twitter::CATEGORY_TERMS_KEY => '(aaa OR bbb) (has:geo OR has:profile_geo)'
}
]
output = twitter_datasource.send :build_queries_from_fields, input_terms
output.should eq expected_output_terms
end
it 'tests search term cut if too many' do
user_mock = CartoDB::Datasources::Doubles::User.new
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
input_terms = {
categories: [
{
category: 'Category 1',
terms: Array(1..35)
}
]
}
expected_output_terms = [
{
Search::Twitter::CATEGORY_NAME_KEY => 'Category 1',
Search::Twitter::CATEGORY_TERMS_KEY => '(1 OR 2 OR 3 OR 4 OR 5 OR 6 OR 7 OR 8 OR 9 OR 10 OR 11 OR 12 OR 13 OR 14 OR 15 OR 16 OR 17 OR 18 OR 19 OR 20 OR 21 OR 22 OR 23 OR 24 OR 25 OR 26 OR 27 OR 28 OR 29) (has:geo OR has:profile_geo)'
},
]
output = twitter_datasource.send :build_queries_from_fields, input_terms
output.should eq expected_output_terms
end
it 'tests search term cut if too big (even if amount is ok)' do
user_mock = CartoDB::Datasources::Doubles::User.new
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
input_terms = {
categories: [
{
category: 'Category 1',
terms: ['wadus1', 'wadus2', 'wadus3' * 500]
}
]
}
expect {
output = twitter_datasource.send :build_queries_from_fields, input_terms
}.to raise_error ParameterError
end
it 'tests date filters' do
user_mock = CartoDB::Datasources::Doubles::User.new
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
input_dates = dates_fixture
output = twitter_datasource.send :build_date_from_fields, input_dates, 'from'
output.should eq '201403031349'
output = twitter_datasource.send :build_date_from_fields, input_dates, 'to'
output.should eq '201403041159'
expect {
twitter_datasource.send :build_date_from_fields, input_dates, 'wadus'
}.to raise_error ParameterError
current_time = Time.now.utc
output = twitter_datasource.send :build_date_from_fields, {
dates: {
toDate: current_time.strftime("%Y-%m-%d"),
toHour: current_time.hour + 1, # Set into the future
toMin: current_time.min
}
}, 'to'
output.should eq nil
end
it 'tests twitter search integration (without conversion to CSV)' do
# This test bridges lots of internal calls to simulate only up until twitter search call and results
user_mock = CartoDB::Datasources::Doubles::User.new
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
input_terms = terms_fixture
input_dates = dates_fixture
Typhoeus.stub(/fakeurl\.carto/) do |request|
accept = (request.options[:headers]||{})['Accept'] || 'application/json'
format = accept.split(',').first
if request.options[:params][:next].nil?
body = data_from_file('sample_tweets.json')
else
body = data_from_file('sample_tweets_2.json')
end
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => format },
body: body
)
end
twitter_api_config = twitter_datasource.send :search_api_config
twitter_api = CartoDB::TwitterSearch::SearchAPI.new(twitter_api_config)
fields = {
categories: input_terms[:categories],
dates: input_dates[:dates]
}
filters = {
Search::Twitter::FILTER_CATEGORIES => (twitter_datasource.send :build_queries_from_fields, fields),
Search::Twitter::FILTER_FROMDATE => (twitter_datasource.send :build_date_from_fields, fields, 'from'),
Search::Twitter::FILTER_TODATE => (twitter_datasource.send :build_date_from_fields, fields, 'to'),
Search::Twitter::FILTER_MAXRESULTS => 500,
Search::Twitter::FILTER_TOTAL_RESULTS => Search::Twitter::NO_TOTAL_RESULTS
}
category = {
name: input_terms[:categories].first[:category],
terms: input_terms[:categories].first[:terms],
}
csv_dumper = twitter_datasource.send :csv_dumper
csv_dumper.begin_dump(input_terms[:categories][0][:category])
csv_dumper.begin_dump(input_terms[:categories][1][:category])
csv_dumper.additional_fields = { category[:name] => category }
output = twitter_datasource.send :search_by_category, twitter_api, filters, category
# 2 pages of 10 results per category search
output.should eq 20
csv_dumper.send :destroy_files
end
it 'tests stopping search if runs out of quota' do
# Should equal to sample_tweets_3.json number of results, and always >= 10 (because is Gnip's minimum)
remaining_tweets_quota = 11
user_mock = CartoDB::Datasources::Doubles::User.new(
twitter_datasource_quota: remaining_tweets_quota
)
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
input_terms = terms_fixture
input_dates = dates_fixture
Typhoeus.stub(/fakeurl\.carto/) do |request|
accept = (request.options[:headers]||{})['Accept'] || 'application/json'
format = accept.split(',').first
if request.options[:params][:next].nil?
# This dataset has 11 items and a "next"
body = data_from_file('sample_tweets_3.json')
else
body = data_from_file('sample_tweets_2.json')
end
Typhoeus::Response.new(
code: 200,
headers: { 'Content-Type' => format },
body: body
)
end
twitter_api_config = twitter_datasource.send :search_api_config
twitter_api = CartoDB::TwitterSearch::SearchAPI.new(twitter_api_config)
fields = {
categories: input_terms[:categories],
dates: input_dates[:dates]
}
filters = {
Search::Twitter::FILTER_CATEGORIES => (twitter_datasource.send :build_queries_from_fields, fields),
Search::Twitter::FILTER_FROMDATE => (twitter_datasource.send :build_date_from_fields, fields, 'from'),
Search::Twitter::FILTER_TODATE => (twitter_datasource.send :build_date_from_fields, fields, 'to'),
Search::Twitter::FILTER_MAXRESULTS => 500,
Search::Twitter::FILTER_TOTAL_RESULTS => Search::Twitter::NO_TOTAL_RESULTS
}
category = {
name: input_terms[:categories].first[:category],
terms: input_terms[:categories].first[:terms],
}
csv_dumper = twitter_datasource.send :csv_dumper
csv_dumper.begin_dump(input_terms[:categories][0][:category])
csv_dumper.begin_dump(input_terms[:categories][1][:category])
csv_dumper.additional_fields = { category[:name] => category }
output = twitter_datasource.send :search_by_category, twitter_api, filters, category
output.should eq remaining_tweets_quota
csv_dumper.send :destroy_files
end
it 'tests user limits on datasource usage' do
user_mock = CartoDB::Datasources::Doubles::User.new
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
# Service enabled tests
result = twitter_datasource.send :is_service_enabled?, CartoDB::Datasources::Doubles::User.new({
has_org: true,
twitter_datasource_enabled: true,
org_twitter_datasource_enabled: true
})
result.should eq true
result = twitter_datasource.send :is_service_enabled?, CartoDB::Datasources::Doubles::User.new({
has_org: true,
twitter_datasource_enabled: false,
org_twitter_datasource_enabled: true
})
result.should eq false
result = twitter_datasource.send :is_service_enabled?, CartoDB::Datasources::Doubles::User.new({
twitter_datasource_enabled: true,
})
result.should eq true
result = twitter_datasource.send :is_service_enabled?, CartoDB::Datasources::Doubles::User.new({
twitter_datasource_enabled: false,
})
result.should eq false
# Quota & soft limit tests
result = twitter_datasource.send :has_enough_quota?, CartoDB::Datasources::Doubles::User.new({
soft_twitter_datasource_limit: false,
twitter_datasource_quota: 10,
})
result.should eq true
result = twitter_datasource.send :has_enough_quota?, CartoDB::Datasources::Doubles::User.new({
soft_twitter_datasource_limit: true,
twitter_datasource_quota: 0,
})
result.should eq true
result = twitter_datasource.send :has_enough_quota?, CartoDB::Datasources::Doubles::User.new({
soft_twitter_datasource_limit: false,
twitter_datasource_quota: 0,
})
result.should eq false
end
it 'checks terms sanitize method' do
user_mock = CartoDB::Datasources::Doubles::User.new
twitter_datasource = Search::Twitter.get_new(get_config, user_mock)
terms = [ 'a', ' b', 'c ', ' d ', ' e f', 'g h ', ' i j ', ' 1 2 3 4 ', ' ' ]
terms_expected = [ 'a', 'b', 'c', 'd', '"e f"', '"g h"', '"i j"', '"1 2 3 4"' ]
result = twitter_datasource.send :sanitize_terms, terms
result.should eq terms_expected
end
end
protected
def terms_fixture
{
categories: [
{
category: 'Category 1',
terms: ['uno', '@dos', '#tres']
},
{
category: 'Category 2',
terms: ['aaa', 'bbb']
}
]
}
end
def dates_fixture
{
dates: {
fromDate: '2014-03-03',
fromHour: '13',
fromMin: '49',
toDate: '2014-03-04',
toHour: '11',
toMin: '59'
}
}
end
def data_from_file(filename, fullpath=false)
if fullpath
File.read(filename)
else
File.read(File.join(File.dirname(__FILE__), "../fixtures/#{filename}"))
end
end
end
+16
View File
@@ -0,0 +1,16 @@
require 'singleton'
module CartoDB
# A facility that abstracts clients from config and also allow for easy injection
class GeocoderConfig
include Singleton
def set(config = {})
@config = config
end
def get()
@config ||= ::Cartodb.config[:geocoder]
end
end
end
@@ -0,0 +1,320 @@
require 'open3'
require 'nokogiri'
require 'csv'
require 'active_support/core_ext/numeric'
require_relative '../../../lib/carto/http/client'
require_relative 'hires_geocoder_interface'
require_relative 'geocoder_config'
module CartoDB
class HiresBatchGeocoder < HiresGeocoderInterface
DEFAULT_TIMEOUT = 5.hours
POLLING_SLEEP_TIME = 5.seconds
LOGGING_TIME = 5.minutes
DOWNLOAD_RETRIES = 5
DOWLOAD_RETRY_SLEEP = 5.seconds
# Generous timeouts, overriden for big files upload/download
HTTP_CONNECTION_TIMEOUT = 60
HTTP_REQUEST_TIMEOUT = 600
# Options for the csv upload endpoint of the Batch Geocoder API
UPLOAD_OPTIONS = {
action: 'run',
indelim: ',',
outdelim: ',',
header: false,
outputCombined: false,
outcols: "displayLatitude,displayLongitude"
}
# INFO: the request_id is the most important thing to care for batch requests
# INFO: it is called remote_id in upper layers
attr_reader :base_url, :request_id, :app_id, :token, :mailto,
:status, :processed_rows, :processed_rows, :successful_processed_rows, :failed_processed_rows,
:empty_processed_rows, :total_rows, :dir, :input_file
class ServiceDisabled < StandardError; end
def initialize(input_csv_file, working_dir, log, geocoding_model)
@input_file = input_csv_file
@dir = working_dir
@log = log
@geocoding_model = geocoding_model
@base_url = config.fetch('base_url')
@app_id = config.fetch('app_id')
@token = config.fetch('token')
@mailto = config.fetch('mailto')
@used_batch_request = true
begin
@batch_api_disabled = config['batch_api_disabled'] == true
rescue
@batch_api_disabled = false
end
end
def run
init_rows_count
@log.append_and_store "Started batched Here geocoding job"
@started_at = Time.now
change_status('running')
upload
# INFO: this loop polls for the state of the table_geocoder batch process
update_status
until ['completed', 'cancelled'].include? @geocoding_model.state do
if timeout?
begin
change_status('timeout')
cancel
ensure
@log.append_and_store "Proceding to cancel job due timeout"
end
end
break if ['failed', 'timeout'].include? @geocoding_model.state
sleep polling_sleep_time
# We don't want to change the status if the job has been cancelled by the user
update_status
update_log_stats
end
update_status
update_log_stats
change_status('completed')
@log.append_and_store "Geocoding Hires job has finished"
ensure
# Processed data at the end of the job
update_status
update_log_stats(false)
end
def upload
assert_batch_api_enabled
@used_batch_request = true
response = http_client.post(
api_url(UPLOAD_OPTIONS),
body: File.open(input_file, "r").read,
headers: { "Content-Type" => "text/plain" },
timeout: 5.hours # more than generous timeout for big file upload
)
handle_api_error(response)
@request_id = extract_response_field(response.body, '//Response/MetaInfo/RequestId')
# TODO: this is a critical error, deal with it appropriately
raise 'Could not get the request ID' unless @request_id
# Update geocodings model with needed data
@geocoding_model.remote_id = @request_id
@geocoding_model.batched = true
@geocoding_model.save
@log.append_and_store "Job sent to HERE, job id: #{@request_id}"
@request_id
end
def used_batch_request?
@used_batch_request
end
def cancel
if @geocoding_model.remote_id.nil?
@log.append_and_store "Can't cancel a HERE geocoder job without the request id"
else
@log.append_and_store "Trying to cancel a batch job sent to HERE"
assert_batch_api_enabled
response = http_client.put(api_url(action: 'cancel'),
connecttimeout: HTTP_CONNECTION_TIMEOUT,
timeout: HTTP_REQUEST_TIMEOUT)
if is_cancellable?(response)
@log.append_and_store "Job was already cancelled"
else
handle_api_error(response)
update_stats(response)
@log.append_and_store "Job sent to HERE has been cancelled"
end
change_status('cancelled')
end
end
def update_status
assert_batch_api_enabled
response = http_client.get(api_url(action: 'status'),
connecttimeout: HTTP_CONNECTION_TIMEOUT,
timeout: HTTP_REQUEST_TIMEOUT)
handle_api_error(response)
update_stats(response)
end
def assert_batch_api_enabled
raise ServiceDisabled if @batch_api_disabled
end
def result
return @result unless @result.nil?
raise 'No request_id provided' unless @geocoding_model.remote_id
results_filename = File.join(dir, "#{@geocoding_model.remote_id}.zip")
download_url = api_url({}, 'result')
download_status_code = nil
retries = 0
while true
if(!download_status_code.nil? && download_status_code == 200)
break
elsif !download_status_code.nil? && download_status_code == 404
# 404 means that the results file is not ready yet
sleep DOWLOAD_RETRY_SLEEP
retries += 1
elsif retries >= DOWNLOAD_RETRIES
raise 'Download request failed: Too many retries, should be a problem with HERE servers'
elsif !download_status_code.nil? && download_status_code > 200 && download_status_code != 404
raise "Download request failed: Http status code #{download_status_code}"
end
download_status_code = execute_results_request(download_url, results_filename)
end
@result = results_filename
end
private
def execute_results_request(download_url, results_filename)
download_status_code = nil
# generous timeout for download of results
request = http_client.request(download_url,
method: :get,
timeout: 5.hours)
File.open(results_filename, 'wb') do |download_file|
request.on_headers do |response|
download_status_code = response.response_code
end
request.on_body do |chunk|
if download_status_code == 200
download_file.write(chunk)
end
end
request.on_complete do |response|
download_status_code = response.response_code
end
request.run
end
return download_status_code
end
def config
GeocoderConfig.instance.get
end
def http_client
@http_client ||= Carto::Http::Client.get('hires_batch_geocoder',
log_requests: true)
end
def api_url(arguments, extra_components = nil)
arguments.merge!(app_id: app_id, token: token, mailto: mailto)
components = [base_url]
# We use the persisted remote_id because we don't have request_id
# in the cancel case due is an instance variable
components << @geocoding_model.remote_id unless @geocoding_model.remote_id.nil?
components << extra_components unless extra_components.nil?
components << '?' + URI.encode_www_form(arguments)
components.join('/')
end
def extract_response_field(response, query)
Nokogiri::XML(response).xpath("#{query}").first.content
rescue NoMethodError => e
CartoDB.notify_exception(e)
nil
end
def extract_numeric_response_field(response, query)
value = extract_response_field(response, query)
return nil if value.blank?
Integer(value)
rescue ArgumentError => e
CartoDB.notify_error("Batch geocoder value error", error: e.message, value: value)
nil
end
def handle_api_error(response)
if response.success? == false
message = extract_response_field(response.body, '//Details')
@failed_processed_rows = number_of_input_file_rows if not input_file.nil?
change_status('failed')
raise "Geocoding API communication failure: #{message}"
end
end
def default_timeout
DEFAULT_TIMEOUT
end
def polling_sleep_time
POLLING_SLEEP_TIME
end
def number_of_input_file_rows
stdout, _status = Open3.capture2('wc', '-l', input_file)
stdout.to_i
end
def update_stats(response)
@status = extract_response_field(response.body, '//Response/Status')
change_status(@status)
@processed_rows = extract_numeric_response_field(response.body, '//Response/ProcessedCount')
@successful_processed_rows = extract_numeric_response_field(response.body, '//Response/SuccessCount')
# addresses that could not be matched
@empty_processed_rows = extract_numeric_response_field(response.body, '//Response/ErrorCount')
# invalid input that could not be processed
@failed_processed_rows = extract_numeric_response_field(response.body, '//Response/InvalidCount')
@total_rows = extract_numeric_response_field(response.body, '//Response/TotalCount')
end
def init_rows_count
@processed_rows = 0
@successful_processed_rows = 0
@empty_processed_rows = 0
@failed_processed_rows = 0
@total_rows = 0
end
def update_log_stats(spaced_by_time=true)
@last_logging_time ||= Time.now
# We don't want to log every few seconds because this kind
# of jobs could last for hours
if (not spaced_by_time) || (Time.now - @last_logging_time) > LOGGING_TIME
@log.append_and_store "Geocoding job status update. "\
"Status: #{@geocoding_model.state} --- Processed rows: #{@processed_rows} "\
"--- Success: #{@successful_processed_rows} --- Empty: #{@empty_processed_rows} "\
"--- Failed: #{@failed_processed_rows}"
@last_logging_time = Time.now
end
end
def timeout?
(Time.now - @started_at) > default_timeout
end
def change_status(status)
@status = status
# The cancelled status should prevail to abort the job
@geocoding_model.refresh
if status != @geocoding_model.state && (not (@geocoding_model.cancelled? || @geocoding_model.timeout?))
@geocoding_model.state = status
@geocoding_model.save
end
end
def is_cancellable?(response)
message = extract_response_field(response.body, '//Details')
response.response_code == 400 && message =~ /CANNOT CANCEL THE COMPLETED, DELETED, FAILED OR ALREADY CANCELLED JOB/
end
end
end
+169
View File
@@ -0,0 +1,169 @@
require 'csv'
require 'json'
require 'open3'
require_relative '../../../lib/carto/http/client'
require_relative 'hires_geocoder_interface'
require_relative 'geocoder_config'
module CartoDB
class HiresGeocoder < HiresGeocoderInterface
# Generous timeouts for this
HTTP_CONNECTION_TIMEOUT = 60
HTTP_REQUEST_TIMEOUT = 600
# Default options for the regular HERE Geocoding API
# Refer to developer.here.com for further reading
GEOCODER_OPTIONS = {
gen: 4, # enables or disables backward incompatible behavior in the API
jsonattributes: 1, # lowercase the first character of each JSON response attribute name
language: 'en-US', # preferred language of address elements in the result
maxresults: 1
}
attr_reader :app_id, :token, :mailto,
:status, :processed_rows, :total_rows, :successful_processed_rows, :failed_processed_rows,
:empty_processed_rows, :dir, :non_batch_base_url
attr_accessor :input_file
def initialize(input_csv_file, working_dir, log, geocoding_model)
@input_file = input_csv_file
@dir = working_dir
@log = log
@geocoding_model = geocoding_model
@non_batch_base_url = config.fetch('non_batch_base_url')
@app_id = config.fetch('app_id')
@token = config.fetch('token')
@mailto = config.fetch('mailto')
init_rows_count
end
def run
init_rows_count
@log.append_and_store "Initialized non batch Here geocoding job"
@result = File.join(dir, 'generated_csv_out.txt')
change_status('running')
@total_rows = input_rows
@log.append_and_store "Total rows to be processed: #{@total_rows}"
::CSV.open(@result, "wb") do |output_csv_file|
::CSV.foreach(input_file, headers: true) do |input_row|
process_row(input_row, output_csv_file)
end
end
change_status('completed')
update_log_stats
@log.append_and_store "Non-batch Here geocoding job finished"
end
def used_batch_request?
false
end
def cancel; end
def update_status; end
def result
@result
end
def request_id
# INFO: there's no request_id for non-batch geocodings
nil
end
private
def config
GeocoderConfig.instance.get
end
def http_client
@http_client ||= Carto::Http::Client.get('hires_geocoder',
log_requests: true)
end
def input_rows
stdout, _stderr, _status = Open3.capture3('wc', '-l', input_file)
stdout.to_i
rescue
0
end
def process_row(input_row, output_csv_file)
@processed_rows += 1
latitude, longitude = geocode_text(input_row["searchtext"])
if !(latitude.nil? || latitude == "") && !(longitude.nil? || longitude == "")
@successful_processed_rows += 1
output_csv_file.add_row [input_row["searchtext"], 1, 1, latitude, longitude]
else
@empty_processed_rows += 1
end
rescue => e
@log.append_and_store "Error processing row with search text #{input_row['searchtext']}: #{e.message}"
CartoDB.notify_debug("Hires geocoding process row error",
error: e.backtrace.join("\n"),
searchtext: input_row["searchtext"],
backtrace: e.backtrace)
@failed_processed_rows += 1
end
def geocode_text(text)
options = GEOCODER_OPTIONS.merge(searchtext: text, app_id: app_id, app_code: token)
url = "#{non_batch_base_url}?#{URI.encode_www_form(options)}"
http_response = http_client.get(url,
connecttimeout: HTTP_CONNECTION_TIMEOUT,
timeout: HTTP_REQUEST_TIMEOUT)
if http_response.success?
response = ::JSON.parse(http_response.body)["response"]
if response['view'].empty?
# no location info for the text input, stop here
return [nil, nil]
end
position = response["view"][0]["result"][0]["location"]["displayPosition"]
return position["latitude"], position["longitude"]
else
CartoDB.notify_debug('Non-batched geocoder failed request', http_response)
return [nil, nil]
end
rescue NoMethodError => e
if e.message == %Q(undefined method `[]' for nil:NilClass)
CartoDB.notify_debug("Non-batched geocoder couldn't parse response",
error: e.backtrace.join("\n"), backtrace: e.backtrace, text: text, response_body: http_response.body)
[nil, nil]
else
raise e
end
end
def api_url(arguments, extra_components = nil)
arguments.merge!(app_id: app_id, token: token, mailto: mailto)
components = [base_url]
components << extra_components unless extra_components.nil?
components << '?' + URI.encode_www_form(arguments)
components.join('/')
end
def init_rows_count
@processed_rows = 0
@successful_processed_rows = 0
@failed_processed_rows = 0
@empty_processed_rows = 0
end
def update_log_stats
@log.append_and_store "Geocoding non-batch Here job status update. "\
"Status: #{@status} --- Processed rows: #{@processed_rows} "\
"--- Success: #{@successful_processed_rows} --- Empty: #{@empty_processed_rows} "\
"--- Failed: #{@failed_processed_rows}"
end
def change_status(status)
@status = status
@geocoding_model.state = status
@geocoding_model.save
end
end
end
@@ -0,0 +1,53 @@
require_relative 'hires_geocoder'
require_relative 'hires_batch_geocoder'
require_relative 'geocoder_config'
module CartoDB
class HiresGeocoderFactory
BATCH_FILES_OVER = 1100 # Use Here Batch Geocoder API with tables over x rows
def self.get(input_csv_file, working_dir, log, geocoding_model, number_of_rows = 0)
geocoder_class = nil
if use_batch_process?(input_csv_file, geocoding_model, number_of_rows)
geocoder_class = HiresBatchGeocoder
else
geocoder_class = HiresGeocoder
end
geocoder_class.new(input_csv_file, working_dir, log, geocoding_model)
end
private
def self.use_batch_process?(input_csv_file, geocoding_model, number_of_rows)
# Due we could check this condition to create the geocoder class and we don't
# have finished yet the csv file generation, and could be nil, we have to check
# multiples conditions. It's sorted by priority
if force_batch? || geocoding_model.batched
true
elsif (not input_csv_file.nil?) && (input_rows(input_csv_file) > BATCH_FILES_OVER)
true
elsif (not number_of_rows.nil?) && (number_of_rows > BATCH_FILES_OVER)
true
else
false
end
end
def self.force_batch?
GeocoderConfig.instance.get['force_batch'] || false
end
def self.input_rows(input_csv_file)
stdout, _stderr, _status = Open3.capture3('wc', '-l', input_csv_file)
stdout.to_i
rescue => e
CartoDB.notify_exception(e)
0
end
end
end
@@ -0,0 +1,11 @@
module CartoDB
class HiresGeocoderInterface
def run
raise 'Not implemented'
end
def cancel
raise 'Not implemented'
end
end
end
+2
View File
@@ -0,0 +1,2 @@
recid,searchtext
"Fredericton, Canada","Fredericton, Canada"
1 recid searchtext
2 Fredericton, Canada Fredericton, Canada
+1
View File
@@ -0,0 +1 @@
<?xml version="1.0" encoding="UTF-8" standalone="yes"?><ns2:SearchBatch xmlns:ns2="http://www.navteq.com/lbsp/Search-Batch/1"><Response><MetaInfo><RequestId>0TK8XRhsIAxMi0KNbEu58MkApMVWctFE</RequestId></MetaInfo><Status>cancelled</Status><JobStarted>2013-09-15T18:19:49.000Z</JobStarted><JobFinished>2013-09-15T18:19:54.000Z</JobFinished><TotalCount>3</TotalCount><ValidCount>2</ValidCount><InvalidCount>1</InvalidCount><ProcessedCount>2</ProcessedCount><PendingCount>0</PendingCount><SuccessCount>2</SuccessCount><ErrorCount>0</ErrorCount></Response></ns2:SearchBatch>
+1
View File
@@ -0,0 +1 @@
<?xml version="1.0" encoding="UTF-8" standalone="yes"?><ns2:SearchBatch xmlns:ns2="http://www.navteq.com/lbsp/Search-Batch/1"><Response><MetaInfo><RequestId>K8DmCWzsZGh4gbawxOuMv2BUcZsIkt7v</RequestId></MetaInfo><Status>submitted</Status><TotalCount>0</TotalCount><ValidCount>0</ValidCount><InvalidCount>0</InvalidCount><ProcessedCount>0</ProcessedCount><PendingCount>0</PendingCount><SuccessCount>0</SuccessCount><ErrorCount>0</ErrorCount></Response></ns2:SearchBatch>
@@ -0,0 +1 @@
{"response":{"metaInfo":{"timestamp":"2014-02-19T12:29:49.723+0000"},"view":[{"result":[{"relevance":1.0,"matchLevel":"country","matchQuality":{"country":1.0},"location":{"locationId":"AREA_21000001","locationType":"point","displayPosition":{"latitude":38.89037,"longitude":-77.03196},"navigationPosition":[{"latitude":38.89037,"longitude":-77.03196}],"mapView":{"topLeft":{"latitude":49.3845,"longitude":-124.749},"bottomRight":{"latitude":24.5018,"longitude":-66.9406}},"address":{"label":"United States","country":"USA","additionalData":[{"value":"United States","key":"CountryName"}]}}}],"viewId":0}]}}
+1
View File
@@ -0,0 +1 @@
<?xml version="1.0" encoding="UTF-8" standalone="yes"?><ns2:Error xmlns:ns2="http://www.navteq.com/lbsp/Errors/1" type="ApplicationError" subtype="InvalidInputData"><Details>Input parameter validation failed. JobId: 9rFyj7kbGMmpF50ZUFAkRnroEiOpDOEZ Email Address is missing!</Details><AdditionalData key="mailto"/></ns2:Error>
+1
View File
@@ -0,0 +1 @@
<?xml version="1.0" encoding="UTF-8" standalone="yes"?><ns2:SearchBatch xmlns:ns2="http://www.navteq.com/lbsp/Search-Batch/1"><Response><MetaInfo><RequestId>0TK8XRhsIAxMi0KNbEu58MkApMVWctFE</RequestId></MetaInfo><Status>completed</Status><JobStarted>2013-09-15T18:19:49.000Z</JobStarted><JobFinished>2013-09-15T18:19:54.000Z</JobFinished><TotalCount>3</TotalCount><ValidCount>2</ValidCount><InvalidCount>1</InvalidCount><ProcessedCount>2</ProcessedCount><PendingCount>0</PendingCount><SuccessCount>2</SuccessCount><ErrorCount>0</ErrorCount></Response></ns2:SearchBatch>
+4
View File
@@ -0,0 +1,4 @@
recId,searchText,country
1,425 W Randolph St, Chicago Illinois 60606,USA
2,31 St James Ave Boston MA 02116,USA
3,10115 Berlin Invalidenstrasse 117,DEU
1 recId,searchText,country
2 1,425 W Randolph St, Chicago Illinois 60606,USA
3 2,31 St James Ave Boston MA 02116,USA
4 3,10115 Berlin Invalidenstrasse 117,DEU
+4
View File
@@ -0,0 +1,4 @@
recId,searchText
1,425 W Randolph St, Chicago Illinois 60606
2,31 St James Ave Boston MA 02116
3,10115 Berlin Invalidenstrasse 117
1 recId,searchText
2 1,425 W Randolph St, Chicago Illinois 60606
3 2,31 St James Ave Boston MA 02116
4 3,10115 Berlin Invalidenstrasse 117
+171
View File
@@ -0,0 +1,171 @@
require_relative '../../../spec/spec_helper'
require_relative '../../../spec/rspec_configuration.rb'
require_relative '../lib/hires_batch_geocoder'
# TODO rename to hires_batch_geocoder_spec.rb or split into batch/non-batch
describe CartoDB::HiresBatchGeocoder do
before(:each) do
@log = mock
@log.stubs(:append)
@log.stubs(:append_and_store)
CartoDB::HiresBatchGeocoder.any_instance.stubs(:config).returns({
'base_url' => 'http://wadus.nokia.com',
'app_id' => '',
'token' => '',
'mailto' => ''
})
@working_dir = Dir.mktmpdir
@geocoding_model = FactoryGirl.create(:geocoding, kind: 'high-resolution', formatter: '{street}',
remote_id: 'wadus')
end
after(:each) do
FileUtils.remove_entry_secure @working_dir
end
describe '#upload' do
it 'returns rec_id on success' do
stub_api_request 200, 'response_example.xml'
filepath = path_to 'without_country.csv'
rec_id = CartoDB::HiresBatchGeocoder.new(filepath, @working_dir, @log, @geocoding_model).upload
rec_id.should eq "K8DmCWzsZGh4gbawxOuMv2BUcZsIkt7v"
end
it 'raises error on failure' do
stub_api_request 400, 'response_failure.xml'
filepath = path_to 'without_country.csv'
expect {
CartoDB::HiresBatchGeocoder.new(filepath, @working_dir, @log, @geocoding_model).upload
}.to raise_error('Geocoding API communication failure: Input parameter validation failed. JobId: 9rFyj7kbGMmpF50ZUFAkRnroEiOpDOEZ Email Address is missing!')
end
end
describe '#update_status' do
before {
stub_api_request(200, 'response_status.xml')
CartoDB::HiresBatchGeocoder.any_instance.stubs(:request_id).returns('wadus')
}
let(:geocoder) { CartoDB::HiresBatchGeocoder.new('/tmp/dummy_input_file.csv', @working_dir, @log, @geocoding_model) }
it "updates status" do
expect { geocoder.update_status }.to change(geocoder, :status).from(nil).to('completed')
end
it "updates processed rows" do
expect { geocoder.update_status }.to change(geocoder, :processed_rows).from(nil).to(2)
end
it "updates total rows" do
expect { geocoder.update_status }.to change(geocoder, :total_rows).from(nil).to(3)
end
end
describe '#result' do
it "saves result file on working directory" do
pending 'move to non-batch suite' # TODO
filepath = path_to 'without_country.csv'
stub_api_request 200, 'response_example_non_batch.json'
geocoder = CartoDB::Geocoder.new(default_params.merge(input_file: filepath, force_batch: false))
geocoder.upload
geocoder.status.should eq 'completed'
result_file = geocoder.result
File.file?(result_file).should be true
File.dirname(result_file).should eq geocoder.dir
end
end
describe '#cancel' do
before {
stub_api_request(200, 'response_cancel.xml')
@geocoding_model.remote_id = 'wadus'
@geocoding_model.save
CartoDB::HiresBatchGeocoder.any_instance.stubs(:request_id).returns('wadus')
}
let(:geocoder) { CartoDB::HiresBatchGeocoder.new('dummy_input_file.csv', @working_dir, @log, @geocoding_model) }
it "updates the status" do
geocoder.cancel
@geocoding_model.state.should eq 'cancelled'
end
end
describe '#extract_response_field' do
let(:geocoder) { CartoDB::HiresBatchGeocoder.new('dummy_input.csv', @working_dir, @log, @geocoding_model) }
let(:response) { File.open(path_to('response_example.xml')).read }
it 'returns specified element value' do
geocoder.send(:extract_response_field, response, '//Response/Status').should == 'submitted'
end
it 'returns nil for missing elements' do
CartoDB.expects(:notify_exception).once
geocoder.send(:extract_response_field, response, 'MissingField').should == nil
end
end
describe '#api_url' do
# TODO move to common place for both geocoders
before(:each) {
CartoDB::HiresBatchGeocoder.any_instance.stubs(:config).returns({
'base_url' => '',
'app_id' => 'a',
'token' => 'b',
'mailto' => 'c'
})
@geocoder = CartoDB::HiresBatchGeocoder.new('dummy_input.csv', @working_dir, @log, @geocoding_model)
}
it 'returns base url by default' do
@geocoder.send(:api_url, {}).should == "/wadus/?app_id=a&token=b&mailto=c"
end
it 'allows for api method specification' do
@geocoder.send(:api_url, {}, 'all').should == "/wadus/all/?app_id=a&token=b&mailto=c"
end
it 'allows for api attributes specification' do
@geocoder.send(:api_url, {attr: 'wadus'}, 'all').should == "/wadus/all/?attr=wadus&app_id=a&token=b&mailto=c"
end
end
describe '#geocode_text' do
it 'returns lat/lon on success' do
pending 'move to non-batched suite' # TODO
stub_api_request 200, 'response_example_non_batch.json'
g = CartoDB::Geocoder.new(default_params)
g.geocode_text("United States").should eq [38.89037, -77.03196]
end
end
describe '#used_batch_request?' do
it 'returns true if sent a request to hi-res batch api' do
pending 'move these to the factory tests' # TODO
stub_api_request 200, 'response_example.xml'
filepath = path_to 'without_country.csv'
geocoder = CartoDB::Geocoder.new(default_params.merge(input_file: filepath))
geocoder.used_batch_request?.should eq true
end
it 'returns false if sent the request was non-batched' do
pending 'move these to the factory tests' # TODO
stub_api_request 200, 'response_example_non_batch.json'
filepath = path_to 'without_country.csv'
g = CartoDB::Geocoder.new(default_params.merge(force_batch: false, input_file: filepath))
g.upload
g.used_batch_request?.should eq false
end
end
def path_to(filepath)
File.expand_path(
File.join(File.dirname(__FILE__), "../spec/fixtures/#{filepath}")
)
end #path_to
def stub_api_request(code, response_file)
response = File.open(path_to(response_file)).read
Typhoeus.stub(/.*nokia.com/).and_return(
Typhoeus::Response.new(code: code, body: response)
)
end
end # CartoDB::Geocoder
@@ -0,0 +1,240 @@
require 'tmpdir'
require 'fileutils'
require_relative '../../../spec/rspec_configuration.rb'
require_relative '../../../spec/spec_helper.rb'
require_relative '../lib/hires_batch_geocoder'
describe CartoDB::HiresBatchGeocoder do
RSpec.configure do |config|
config.before :each do
Typhoeus::Expectation.clear
end
end
before(:each) do
@log = mock
@log.stubs(:append)
@log.stubs(:append_and_store)
@working_dir = Dir.mktmpdir
@input_csv_file = path_to '../../table-geocoder/spec/fixtures/nokia_input.csv'
CartoDB::HiresBatchGeocoder.any_instance.stubs(:config).returns({
'base_url' => 'batch.example.com',
'app_id' => '',
'token' => '',
'mailto' => ''
})
@geocoding_model = FactoryGirl.create(:geocoding, kind: 'high-resolution', formatter: '{street}' )
@batch_geocoder = CartoDB::HiresBatchGeocoder.new(@input_csv_file, @working_dir, @log, @geocoding_model)
end
after(:each) do
FileUtils.rm_f @working_dir
end
describe '#run' do
it 'uploads a file to the batch server' do
mock_complete_response
@batch_geocoder.expects(:upload).once
@batch_geocoder.run
@geocoding_model.state.should == 'completed'
end
it 'times out if not finished before the DEFAULT_TIMEOUT' do
mock_complete_response('running')
@batch_geocoder.expects(:upload).once
@batch_geocoder.expects(:cancel).once
@batch_geocoder.stubs(:default_timeout).returns(-10) # make sure it times out
@batch_geocoder.run
@geocoding_model.state.should == 'timeout'
end
end
describe '#upload' do
it 'uploads a file to the batch service' do
url = @batch_geocoder.send(:api_url, CartoDB::HiresBatchGeocoder::UPLOAD_OPTIONS)
expected_request_id = 'dummy_id'
xml_response_body = "<Response><MetaInfo><RequestId>#{expected_request_id}</RequestId></MetaInfo></Response>"
response = Typhoeus::Response.new(code: 200, body: xml_response_body)
Typhoeus.stub(url, method: :post).and_return(response)
@batch_geocoder.upload
@batch_geocoder.request_id.should == expected_request_id
@batch_geocoder.used_batch_request?.should == true
end
it 'raises an exception if the api returns != 200' do
url = @batch_geocoder.send(:api_url, CartoDB::HiresBatchGeocoder::UPLOAD_OPTIONS)
response = Typhoeus::Response.new(code: 401)
Typhoeus.stub(url, method: :post).and_return(response)
CartoDB.expects(:notify_exception).once
expect {
@batch_geocoder.upload
}.to raise_error(RuntimeError, /Geocoding API communication failure/)
end
it 'raises an exception if the api does not return a RequestId' do
url = @batch_geocoder.send(:api_url, CartoDB::HiresBatchGeocoder::UPLOAD_OPTIONS)
response = Typhoeus::Response.new(code: 200)
Typhoeus.stub(url, method: :post).and_return(response)
CartoDB.expects(:notify_exception).once
expect {
@batch_geocoder.upload
}.to raise_error(RuntimeError, /Could not get the request ID/)
end
end
describe '#cancel' do
it 'sends a cancel put request and gets the status, processed and total rows' do
request_id = 'dummy_request_id'
@geocoding_model.remote_id = request_id
@geocoding_model.save
@batch_geocoder.stubs(:request_id).returns(request_id)
url = @batch_geocoder.send(:api_url, action: 'cancel')
url.should match(%r'/#{request_id}/')
url.should match(%r'action=cancel')
expected_status = 'cancelled'
expected_processed_rows = 20
expected_success_rows = 17
expected_failed_rows = 0
expected_empty_rows = 3
expected_total_rows = 30
response_body = <<END_XML
<Response>
<Status>#{expected_status}</Status>
<ProcessedCount>#{expected_processed_rows}</ProcessedCount>
<SuccessCount>#{expected_success_rows}</SuccessCount>
<ErrorCount>#{expected_failed_rows}</ErrorCount>
<InvalidCount>#{expected_empty_rows}</InvalidCount>
<TotalCount>#{expected_total_rows}</TotalCount>
</Response>
END_XML
response = Typhoeus::Response.new(code: 200, body: response_body)
Typhoeus.stub(url, method: :put).and_return(response)
@batch_geocoder.cancel
@batch_geocoder.status.should == expected_status
@batch_geocoder.processed_rows.should == expected_processed_rows
@batch_geocoder.total_rows.should == expected_total_rows
end
end
describe '#update' do
it 'gets the status, processed and total rows by sending a get request' do
request_id = 'dummy_request_id'
@geocoding_model.remote_id = request_id
@geocoding_model.save
@batch_geocoder.stubs(:request_id).returns(request_id)
url = @batch_geocoder.send(:api_url, action: 'status')
url.should match(%r'/#{request_id}/')
url.should match(%r'action=status')
expected_status = 'running'
expected_processed_rows = 20
expected_success_rows = 17
expected_failed_rows = 0
expected_empty_rows = 3
expected_total_rows = 30
response_body = <<END_XML
<Response>
<Status>#{expected_status}</Status>
<ProcessedCount>#{expected_processed_rows}</ProcessedCount>
<SuccessCount>#{expected_success_rows}</SuccessCount>
<ErrorCount>#{expected_failed_rows}</ErrorCount>
<InvalidCount>#{expected_empty_rows}</InvalidCount>
<TotalCount>#{expected_total_rows}</TotalCount>
</Response>
END_XML
response = Typhoeus::Response.new(code: 200, body: response_body)
Typhoeus.stub(url, method: :get).and_return(response)
@batch_geocoder.update_status
@batch_geocoder.status.should == expected_status
@batch_geocoder.processed_rows.should == expected_processed_rows
@batch_geocoder.total_rows.should == expected_total_rows
end
end
describe '#result' do
it "raises an exception if there's no request_id from a previous upload" do
expect {
@batch_geocoder.result
}.to raise_error(RuntimeError, /No request_id provided/)
end
it 'downloads the result file from the remote server' do
request_id = 'dummy_request_id'
@geocoding_model.remote_id = request_id
@geocoding_model.save
@batch_geocoder.stubs(:request_id).returns(request_id)
expected_response_body = 'dummy result file contents'
url = @batch_geocoder.send(:api_url, {}, 'result')
response = Typhoeus::Response.new(code: 200, body: expected_response_body)
Typhoeus.stub(url, method: :get).and_return(response)
result_file = @batch_geocoder.result
File.open(result_file).read.should == expected_response_body
# it also "memoizes" the result file and avoids further downloads
@batch_geocoder.expects(:http_client).never
@batch_geocoder.result.should == result_file
end
it 'raises an exception if cannot get a result file' do
request_id = 'dummy_request_id'
@geocoding_model.remote_id = request_id
@geocoding_model.save
@batch_geocoder.stubs(:request_id).returns(request_id)
expected_response_body = 'dummy result file contents'
url = @batch_geocoder.send(:api_url, {}, 'result')
response = Typhoeus::Response.new(code: 400)
Typhoeus.stub(url, method: :get).and_return(response)
expect {
@batch_geocoder.result
}.to raise_error(RuntimeError, /Download request failed/)
end
end
def path_to(filepath = '')
File.expand_path(
File.join(File.dirname(__FILE__), "../fixtures/#{filepath}")
)
end
def mock_complete_response(state='completed')
@geocoding_model.remote_id = 'dummy_id'
@geocoding_model.save.reload
url = @batch_geocoder.send(:api_url, {action: 'status'})
xml_response_body = '<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
<ns2:SearchBatch xmlns:ns2="http://www.navteq.com/lbsp/Search-Batch/1">
<Response>
<MetaInfo>
<RequestId>dummy_id</RequestId>
</MetaInfo>
<Status>'+state+'</Status>
<JobStarted>2016-04-08T08:24:05.000Z</JobStarted>
<JobFinished>2016-04-08T08:24:39.000Z</JobFinished>
<TotalCount>1</TotalCount>
<ValidCount>1</ValidCount>
<InvalidCount>0</InvalidCount>
<ProcessedCount>1</ProcessedCount>
<PendingCount>0</PendingCount>
<SuccessCount>1</SuccessCount>
<ErrorCount>0</ErrorCount>
</Response>
</ns2:SearchBatch>'
response = Typhoeus::Response.new(code: 200, body: xml_response_body)
Typhoeus.stub(url, method: :get).and_return(response)
end
end
@@ -0,0 +1,68 @@
require_relative '../../../spec/rspec_configuration'
require_relative '../../../spec/spec_helper'
require_relative '../lib/hires_geocoder_factory'
require_relative '../lib/geocoder_config'
describe CartoDB::HiresGeocoderFactory do
after(:all) do
# reset config
CartoDB::GeocoderConfig.instance.set(nil)
end
before(:each) do
@log = mock
@log.stubs(:append)
@log.stubs(:append_and_store)
@geocoding_model = FactoryGirl.create(:geocoding, kind: 'high-resolution', formatter: '{street}')
end
describe '#get' do
it 'returns a HiresGeocoder instance if the input file has less than N rows' do
CartoDB::GeocoderConfig.instance.set({
'non_batch_base_url' => 'http://api.example.com',
'app_id' => 'dummy_app_id',
'token' => 'dummy_token',
'mailto' => 'dummy_mail_addr'
})
dummy_input_file = 'dummy_input_file.csv'
working_dir = '/tmp/any_dir'
input_rows = CartoDB::HiresGeocoderFactory::BATCH_FILES_OVER - 1
CartoDB::HiresGeocoderFactory.expects(:input_rows).once.with(dummy_input_file).returns(input_rows)
CartoDB::HiresGeocoderFactory.get(dummy_input_file, working_dir, @log, @geocoding_model).class.should == CartoDB::HiresGeocoder
end
it 'returns a HiresBatchGeocoder instance if the input file is above N rows' do
CartoDB::GeocoderConfig.instance.set({
'base_url' => 'http://api.example.com',
'app_id' => 'dummy_app_id',
'token' => 'dummy_token',
'mailto' => 'dummy_mail_addr'
})
dummy_input_file = 'dummy_input_file.csv'
working_dir = '/tmp/any_dir'
input_rows = CartoDB::HiresGeocoderFactory::BATCH_FILES_OVER + 1
CartoDB::HiresGeocoderFactory.expects(:input_rows).once.with(dummy_input_file).returns(input_rows)
CartoDB::HiresGeocoderFactory.get(dummy_input_file, working_dir, @log, @geocoding_model).class.should == CartoDB::HiresBatchGeocoder
end
it 'returns a batch geocoder if config has force_batch set to true' do
CartoDB::GeocoderConfig.instance.set({
'force_batch' => true,
'base_url' => 'http://api.example.com',
'app_id' => 'dummy_app_id',
'token' => 'dummy_token',
'mailto' => 'dummy_mail_addr'
})
dummy_input_file = 'dummy_input_file.csv'
working_dir = '/tmp/any_dir'
CartoDB::HiresGeocoderFactory.expects(:input_rows).never
CartoDB::HiresGeocoderFactory.get(dummy_input_file, working_dir, @log, @geocoding_model).class.should == CartoDB::HiresBatchGeocoder
end
end
end
@@ -0,0 +1,135 @@
require 'tmpdir'
require 'fileutils'
require 'csv'
require_relative '../../../spec/rspec_configuration'
require_relative '../../../spec/spec_helper'
require_relative '../lib/hires_geocoder'
describe CartoDB::HiresGeocoder do
MOCK_COORDINATES = [38.89037, -77.03196]
RSpec.configure do |config|
config.before :each do
Typhoeus::Expectation.clear
end
end
before(:each) do
@working_dir = Dir.mktmpdir
@input_csv_file = path_to '../../table-geocoder/spec/fixtures/nokia_input.csv'
@log = mock
@log.stubs(:append)
@log.stubs(:append_and_store)
CartoDB::HiresGeocoder.any_instance.stubs(:config).returns({
'non_batch_base_url' => 'batch.example.com',
'app_id' => '',
'token' => '',
'mailto' => ''
})
@geocoding_model = FactoryGirl.create(:geocoding, kind: 'high-resolution', formatter: '{street}')
@geocoder = CartoDB::HiresGeocoder.new(@input_csv_file, @working_dir, @log, @geocoding_model)
end
after(:each) do
FileUtils.rm_f @working_dir
end
describe '#run' do
it 'takes every row from input and calls geocode_text on them' do
rows_to_geocode = ::CSV.read(@input_csv_file, headers: true).length
@geocoder.expects(:geocode_text).times(rows_to_geocode).returns(MOCK_COORDINATES)
@geocoder.run
@geocoder.status.should == 'completed'
end
end
describe '#process_row' do
it 'increments the number of processed rows by one when called' do
output_csv_mock = mock
output_csv_mock.expects(:add_row).once
input_row = {'searchtext' => 'olakase'}
@geocoder.expects(:geocode_text).once.returns(MOCK_COORDINATES)
@geocoder.processed_rows.should == 0
@geocoder.send(:process_row, input_row, output_csv_mock)
@geocoder.processed_rows.should == 1
end
it 'adds a row with the expected format when success' do
output_csv_mock = mock
output_csv_mock.expects(:add_row).once.with ['olakase', 1, 1, MOCK_COORDINATES[0], MOCK_COORDINATES[1]]
input_row = {'searchtext' => 'olakase'}
@geocoder.expects(:geocode_text).once.returns(MOCK_COORDINATES)
@geocoder.send(:process_row, input_row, output_csv_mock)
end
it 'does not add any row when it when geolocation fails' do
output_csv_mock = mock
output_csv_mock.expects(:add_row).never
input_row = {'searchtext' => 'olakase'}
@geocoder.expects(:geocode_text).once.returns([nil, nil])
@geocoder.send(:process_row, input_row, output_csv_mock)
end
end
describe '#geocode_text' do
it 'sends a request to the non-batched geocoder service and gets a couple of coordinates' do
json_response_body = {
response: {
view: [
result: [
location: {
displayPosition: {
latitude: MOCK_COORDINATES[0],
longitude: MOCK_COORDINATES[1]
}
}
]
]
}
}.to_json
mocked_response = Typhoeus::Response.new(code: 200, body: json_response_body)
Typhoeus.stub(//, method: :get).and_return(mocked_response)
@geocoder.send(:geocode_text, 'Dummy address').should == MOCK_COORDINATES
end
it "returns nil coordinates if the http request doesn't succeed" do
mocked_response = Typhoeus::Response.new(code: 500)
Typhoeus.stub(//, method: :get).and_return(mocked_response)
CartoDB.expects(:notify_debug).with('Non-batched geocoder failed request', mocked_response).once
@geocoder.send(:geocode_text, 'Dummy address').should == [nil, nil]
end
it 'returns nil coordinates and log a trace if it is not able to parse the response' do
input_text = 'Dummy address'
json_response_body = {
unexpected: 'this response body has unexpected format for whatever reason'
}.to_json
mocked_response = Typhoeus::Response.new(code: 200, body: json_response_body)
Typhoeus.stub(//, method: :get).and_return(mocked_response)
CartoDB.expects(:notify_debug).with("Non-batched geocoder couldn't parse response", anything()).once
@geocoder.send(:geocode_text, input_text).should == [nil, nil]
end
it 'returns nil coordinates and stops there if the response does not contain any location' do
input_text = 'Dummy address'
json_response_body = '{"response":{"metaInfo":{"timestamp":"2015-07-14T15:33:35.023+0000"},"view":[]}}'
mocked_response = Typhoeus::Response.new(code: 200, body: json_response_body)
Typhoeus.stub(//, method: :get).and_return(mocked_response)
@geocoder.send(:geocode_text, input_text).should == [nil, nil]
end
end
def path_to(filepath = '')
File.expand_path(
File.join(File.dirname(__FILE__), "../fixtures/#{filepath}")
)
end
end
+21
View File
@@ -0,0 +1,21 @@
guard 'minitest', test_folders: 'spec' do
# with Minitest::Spec
watch(%r|^spec/unit/(.*)_spec\.rb|)
watch(%r|^(.*)\.rb|) { |m| "spec/unit/#{m[1]}_spec.rb" }
# with Minitest::Unit
# watch(%r|^test/(.*)\/?test_(.*)\.rb|)
# watch(%r|^lib/(.*)([^/]+)\.rb|) { |m| "test/#{m[1]}test_#{m[2]}.rb" }
# watch(%r|^test/test_helper\.rb|) { "test" }
# Rails 3.2
# watch(%r|^app/controllers/(.*)\.rb|) { |m| "test/controllers/#{m[1]}_test.rb" }
# watch(%r|^app/helpers/(.*)\.rb|) { |m| "test/helpers/#{m[1]}_test.rb" }
# watch(%r|^app/models/(.*)\.rb|) { |m| "test/unit/#{m[1]}_test.rb" }
# Rails
# watch(%r|^app/controllers/(.*)\.rb|) { |m| "test/functional/#{m[1]}_test.rb" }
# watch(%r|^app/helpers/(.*)\.rb|) { |m| "test/helpers/#{m[1]}_test.rb" }
# watch(%r|^app/models/(.*)\.rb|) { |m| "test/unit/#{m[1]}_test.rb" }
end
+17
View File
@@ -0,0 +1,17 @@
require 'rake/testtask'
Rake::TestTask.new do |t|
t.libs << "test"
t.pattern = "spec/**/*_spec.rb"
end
Rake::TestTask.new('test:unit') do |t|
t.libs << "test"
t.pattern = "spec/unit/**/*_spec.rb"
end
Rake::TestTask.new('test:acceptance') do |t|
t.libs << "test"
t.pattern = "spec/acceptance/**/*_spec.rb"
end
@@ -0,0 +1,21 @@
module CartoDB
module Importer2
module QuotaCheckHelpers
def raise_if_over_storage_quota(requested_quota: 0, available_quota: 0, user_id: nil)
quota_overage = requested_quota - available_quota
if quota_overage > 0
report_over_quota(user_id, quota_overage: quota_overage) if user_id
raise StorageQuotaExceededError.new
end
end
def report_over_quota(user_id, quota_overage: 0)
Carto::Tracking::Events::ExceededQuota.new(user_id,
user_id: user_id,
quota_overage: quota_overage).report
end
end
end
end
+10
View File
@@ -0,0 +1,10 @@
require_relative './importer/column'
require_relative './importer/downloader'
require_relative './importer/datasource_downloader'
require_relative './importer/georeferencer'
require_relative './importer/job'
require_relative './importer/loader'
require_relative './importer/ogr2ogr'
require_relative './importer/runner'
require_relative './importer/source_file'
@@ -0,0 +1,28 @@
module CartoDB
module Importer2
class CartodbfyTime
@@instances = {}
# Gets an instance unique per process + data_import_id
def self.instance(data_import_id)
@@instances[data_import_id] ||= new
end
def initialize
@cartodbfy_time = 0.0
end
def add(elapsed_time)
@cartodbfy_time += elapsed_time
end
def get
return @cartodbfy_time
end
end
end
end
+291
View File
@@ -0,0 +1,291 @@
require 'active_support/time'
require_relative './job'
require_relative './string_sanitizer'
require_relative './exceptions'
require_relative './query_batcher'
module CartoDB
module Importer2
class Column
DEFAULT_SRID = 4326
WKB_RE = /^\d{2}/
GEOJSON_RE = /{.*(type|coordinates).*(type|coordinates).*}/
WKT_RE = /POINT|LINESTRING|POLYGON/
KML_MULTI_RE = /<Line|<Polygon/
KML_POINT_RE = /<Point>/
DEFAULT_SCHEMA = 'cdb_importer'
DIRECT_STATEMENT_TIMEOUT = 1.hour * 1000
# @see config/initializers/carto_db.rb -> POSTGRESQL_RESERVED_WORDS
RESERVED_WORDS = %w{ ALL ANALYSE ANALYZE AND ANY ARRAY AS ASC ASYMMETRIC
AUTHORIZATION BETWEEN BINARY BOTH CASE CAST CHECK
COLLATE COLUMN CONSTRAINT CREATE CROSS CURRENT_DATE
CURRENT_ROLE CURRENT_TIME CURRENT_TIMESTAMP
CURRENT_USER DEFAULT DEFERRABLE DESC DISTINCT DO
ELSE END EXCEPT FALSE FOR FOREIGN FREEZE FROM FULL
GRANT GROUP HAVING ILIKE IN INITIALLY INNER INTERSECT
INTO IS ISNULL JOIN LEADING LEFT LIKE LIMIT LOCALTIME
LOCALTIMESTAMP NATURAL NEW NOT NOTNULL NULL OFF
OFFSET OLD ON ONLY OR ORDER OUTER OVERLAPS PLACING
PRIMARY REFERENCES RIGHT SELECT SESSION_USER SIMILAR
SOME SYMMETRIC TABLE THEN TO TRAILING TRUE UNION
UNIQUE USER USING VERBOSE WHEN WHERE XMIN XMAX
FORMAT CONTROLLER ACTION
}
def initialize(db, table_name, column_name, user, schema = DEFAULT_SCHEMA, job = nil, logger = nil, capture_exceptions = true)
@job = job || Job.new({logger: logger})
@db = db
@table_name = table_name
@column_name = column_name.to_sym
@schema = schema
@capture_exceptions = capture_exceptions
@user = user
@from_geojson_with_transform = false
end
def mark_as_from_geojson_with_transform
@from_geojson_with_transform = true
end
def type
db.schema(table_name, reload: true, schema: schema)
.select { |column_details|
column_details.first == column_name
}.last.last.fetch(:db_type)
end
def geometrify
job.log 'geometrifying'
raise "empty column #{column_name}" if empty?
convert_from_wkt if wkt?
convert_from_kml_multi if kml_multi?
convert_from_kml_point if kml_point?
convert_from_geojson_with_transform if geojson? && @from_geojson_with_transform
convert_from_geojson if geojson?
cast_to('geometry')
convert_to_2d
job.log 'geometrified'
self
end
def convert_from_wkt
#TODO: @capture_exceptions
job.log 'Converting geometry from WKT to WKB'
@user.db_service.in_database_direct_connection(statement_timeout: DIRECT_STATEMENT_TIMEOUT) do |user_direct_conn|
user_direct_conn.run(%Q{
UPDATE #{qualified_table_name}
SET #{column_name} = ST_GeomFromText(#{column_name}, #{DEFAULT_SRID})
})
end
self
end
def convert_from_geojson_with_transform
# 1) cast to proper null the geom column
db.run(%Q{
UPDATE #{qualified_table_name}
SET #{column_name} = NULL
WHERE #{column_name} = ''
})
# 2) Normal geojson behavior
#TODO: @capture_exceptions
job.log 'Converting geometry from GeoJSON with transform to WKB'
@user.db_service.in_database_direct_connection(statement_timeout: DIRECT_STATEMENT_TIMEOUT) do |user_direct_conn|
user_direct_conn.run(%Q{
UPDATE #{qualified_table_name}
SET #{column_name} = public.ST_SetSRID(public.ST_GeomFromGeoJSON(#{column_name}), #{DEFAULT_SRID})
})
end
self
end
def convert_from_geojson
#TODO: @capture_exceptions
job.log 'Converting geometry from GeoJSON to WKB'
@user.db_service.in_database_direct_connection(statement_timeout: DIRECT_STATEMENT_TIMEOUT) do |user_direct_conn|
user_direct_conn.run(%Q{
UPDATE #{qualified_table_name}
SET #{column_name} = public.ST_SetSRID(public.ST_GeomFromGeoJSON(#{column_name}), #{DEFAULT_SRID})
})
end
self
end
def convert_from_kml_point
#TODO: @capture_exceptions
job.log 'Converting geometry from KML point to WKB'
@user.db_service.in_database_direct_connection(statement_timeout: DIRECT_STATEMENT_TIMEOUT) do |user_direct_conn|
user_direct_conn.run(%Q{
UPDATE #{qualified_table_name}
SET #{column_name} = public.ST_SetSRID(public.ST_GeomFromKML(#{column_name}),#{DEFAULT_SRID})
})
end
end
def convert_from_kml_multi
#TODO: @capture_exceptions
job.log 'Converting geometry from KML multi to WKB'
@user.db_service.in_database_direct_connection(statement_timeout: DIRECT_STATEMENT_TIMEOUT) do |user_direct_conn|
user_direct_conn.run(%Q{
UPDATE #{qualified_table_name}
SET #{column_name} = public.ST_SetSRID(public.ST_Multi(public.ST_GeomFromKML(#{column_name})),#{DEFAULT_SRID})
})
end
end
def convert_to_2d
#TODO: @capture_exceptions
job.log 'Converting to 2D point'
@user.db_service.in_database_direct_connection(statement_timeout: DIRECT_STATEMENT_TIMEOUT) do |user_direct_conn|
user_direct_conn.run(%Q{
UPDATE #{qualified_table_name}
SET #{column_name} = public.ST_Force2D(#{column_name})
})
end
end
def wkb?
!!(sample.to_s =~ WKB_RE)
end
def wkt?
!!(sample.to_s =~ WKT_RE)
end
def geojson?
!!(sample.to_s =~ GEOJSON_RE)
end
def kml_point?
!!(sample.to_s =~ KML_POINT_RE)
end
def kml_multi?
!!(sample.to_s =~ KML_MULTI_RE)
end
def cast_to(type)
job.log "casting #{column_name} to #{type}"
@user.db_service.in_database_direct_connection(statement_timeout: DIRECT_STATEMENT_TIMEOUT) do |user_direct_conn|
user_direct_conn.run(%Q{
ALTER TABLE #{qualified_table_name}
ALTER #{column_name}
TYPE #{type}
USING #{column_name}::#{type}
})
end
self
end
def sample
return nil if empty?
records_with_data.first.fetch(column_name)
end
def empty?
records_with_data.empty?
end
def records_with_data
@records_with_data ||= db[%Q{
SELECT #{column_name} FROM "#{schema}"."#{table_name}"
WHERE #{column_name} IS NOT NULL
AND #{column_name}::text != ''
LIMIT 1
}]
end
def rename_to(new_name)
return self if new_name.to_s == column_name.to_s
job.log "Renaming column #{column_name} TO #{new_name}"
db.run(%Q{
ALTER TABLE "#{schema}"."#{table_name}"
RENAME COLUMN "#{column_name}" TO "#{new_name}"
})
@column_name = new_name
end
def geometry_type
sample = db[%Q{
SELECT public.GeometryType(ST_Force2D(#{column_name}::geometry))
AS type
FROM #{schema}.#{table_name}
WHERE #{column_name} IS NOT NULL
LIMIT 1
}].first
sample && sample.fetch(:type)
end
def drop
db.run(%Q{
ALTER TABLE #{qualified_table_name}
DROP COLUMN IF EXISTS #{column_name}
})
end
# Replace empty strings by nulls to avoid cast errors
def empty_lines_to_nulls
# first timeout crash
job.log 'replace empty strings by nulls?'
column_id = column_name.to_sym
column_type = nil
db.schema(table_name).each do |colid, coldef|
if colid == column_id
column_type = coldef[:type]
end
end
if column_type != nil && column_type == :string
#TODO: @capture_exceptions
job.log 'string column found, replacing'
@user.db_service.in_database_direct_connection(statement_timeout: DIRECT_STATEMENT_TIMEOUT) do |user_direct_conn|
user_direct_conn.run(%Q{
UPDATE #{qualified_table_name}
SET #{column_name}=NULL
WHERE #{column_name}=''
})
end
else
job.log 'no string column found, nothing replaced'
end
end
def sanitize
rename_to(sanitized_name)
end
def sanitized_name
name = StringSanitizer.new.sanitize(column_name.to_s)
return name unless reserved?(name) || unsupported?(name)
"_#{name}"
end
def reserved?(name)
RESERVED_WORDS.include?(name.upcase)
end
def unsupported?(name)
name !~ /^[a-zA-Z_]/
end
private
attr_reader :job, :db, :table_name, :column_name, :schema
def qualified_table_name
%Q("#{schema}"."#{table_name}")
end
end
end
end
@@ -0,0 +1,156 @@
require 'carto/connector'
require_relative 'exceptions'
require_relative 'georeferencer'
require_relative 'runner_helper'
module CartoDB
module Importer2
# ConnectorRunner runs connector-based imports to import datasets from external databases.
#
# ConnectorRunner has the same public API as Runner, but
# instead of downloading and unpacking files and then processing them with ogr2ogr to import them into
# a user database table, a Connector instance connects directly to a remote database and imports
# the specified dataset into a table in the user's database. Currently FDW are used for this.
#
# A connector runner is defined by a hash of parameters; the `provider` parameter is always required and specifies
# the Connector class, which in turn defines the rest of valid parameters for the connector.
#
class ConnectorRunner
include CartoDB::Importer2::RunnerHelper
attr_reader :results, :log, :job, :warnings
attr_accessor :stats
def initialize(connector_source, options = {})
@pg_options = options[:pg]
@log = options[:log] || new_logger
@job = options[:job] || new_job(@log, @pg_options)
@user = options[:user]
@collision_strategy = options[:collision_strategy]
@georeferencer = options[:georeferencer] || new_georeferencer(@job)
@id = @job.id
@unique_suffix = @id.delete('-')
@json_params = JSON.parse(connector_source)
extract_params
@connector = Carto::Connector.new(@params, user: @user, logger: @log)
@results = []
@tracker = nil
@stats = {}
@warnings = {}
@connector.check_availability!
end
def run(tracker = nil)
@tracker = tracker
@job.log "ConnectorRunner #{@json_params.except('connection').to_json}"
# TODO: logging with CartoDB::Logger
table_name = @job.table_name
if should_import?(@connector.remote_table_name)
@job.log "Copy connected table"
warnings = @connector.copy_table(schema_name: @job.schema, table_name: @job.table_name)
@job.log 'Georeference geometry column'
georeference
@warnings.merge! warnings if warnings.present?
else
@job.log "Table #{table_name} won't be imported"
end
rescue => error
@job.log "ConnectorRunner Error #{error}"
@results.push result_for(@job.schema, table_name, error)
else
if should_import?(@connector.remote_table_name)
@job.log "ConnectorRunner created table #{table_name}"
@job.log "job schema: #{@job.schema}"
@results.push result_for(@job.schema, table_name)
end
end
def georeference
@georeferencer.run
rescue => error
@job.log "ConnectorRunner Error while georeference #{error}"
end
def remote_data_updated?
@connector.remote_data_updated?
end
def tracker
@tracker || lambda { |state| state }
end
def visualizations
# This method is needed to make the interface of ConnectorRunner compatible with Runner
[]
end
# General availability of connectors for a user
def self.check_availability!(user)
Carto::Connector.check_availability! user
end
def etag
# This method is needed to make the interface of ConnectorRunner compatible with Runner,
# but we have no meaningful data to return here.
end
def checksum
# This method is needed to make the interface of ConnectorRunner compatible with Runner,
# but we have no meaningful data to return here.
end
def last_modified
# This method is needed to make the interface of ConnectorRunner compatible with Runner,
# but we have no meaningful data to return here.
end
def provider_name
@connector.provider_name
end
private
# Parse @json_params and extract @params
def extract_params
@params = Carto::Connector::Parameters.new(@json_params)
end
def result_table_name
Carto::DB::Sanitize.sanitize_identifier @connector.remote_table_name
end
def result_for(schema, table_name, error = nil)
@job.success_status = !error
@job.logger.store
Result.new(
name: result_table_name,
schema: schema,
tables: [table_name],
success: @job.success_status,
error_code: error_for(error),
log_trace: @job.logger.to_s,
support_tables: []
)
end
def new_logger
CartoDB::Log.new(type: CartoDB::Log::TYPE_DATA_IMPORT)
end
def new_job(log, pg_options)
Job.new(logger: log, pg_options: pg_options)
end
def new_georeferencer(job)
Georeferencer.new(job.db, job.table_name, {}, Georeferencer::DEFAULT_SCHEMA, job)
end
UNKNOWN_ERROR_CODE = 99999
def error_for(exception)
exception && ERRORS_MAP.fetch(exception.class, UNKNOWN_ERROR_CODE)
end
end
end
end
@@ -0,0 +1,223 @@
require_relative 'ip_checker'
require_relative 'table_sampler'
require_relative 'namedplaces_guesser'
require_relative '../../../../lib/cartodb/stats/importer'
module CartoDB
module Importer2
class ContentGuesser
SQLAPI_CALLS_TIMEOUT = 45
COUNTRIES_COLUMN = 'name_'
COUNTRIES_QUERY = "SELECT #{COUNTRIES_COLUMN} FROM admin0_synonyms"
DEFAULT_MINIMUM_ENTROPY = 0.9
ID_COLUMNS = ['ogc_fid', 'gid', 'cartodb_id', 'objectid'].freeze
attr_reader :country_name_normalizer
def initialize(db, table_name, schema, options, job=nil)
@db = db
@table_name = table_name
@schema = schema
@options = options
@job = job
@importer_stats = CartoDB::Stats::Importer.instance
@country_name_normalizer = Proc.new {|str| str.nil? ? '' : str.gsub(/[^a-zA-Z\u00C0-\u00ff]+/, '').downcase }
end
def set_importer_stats(importer_stats)
@importer_stats = importer_stats
end
def enabled?
@options[:guessing][:enabled] rescue false
end
def country_column
return nil if not enabled?
columns.each do |column|
return column[:column_name] if is_country_column? column
end
nil
end
def namedplaces
@namedplaces ||= NamedplacesGuesser.new(self)
end
def ip_column
return nil if not enabled?
columns.each do |column|
return column[:column_name] if is_ip_column? column
end
nil
end
def columns
@columns ||= @db[%Q(
SELECT column_name, data_type
FROM information_schema.columns
WHERE table_name = '#{@table_name}' AND table_schema = '#{@schema}'
)]
end
def is_country_column?(column)
return false unless is_text_type? column
entropy = metric_entropy(column, country_name_normalizer)
if entropy < minimum_entropy
false
else
proportion = country_proportion(column)
if proportion < threshold
false
else
log_country_guessing_match_metrics(proportion)
true
end
end
end
def log_country_guessing_match_metrics(proportion)
@importer_stats.gauge('country_proportion', proportion)
end
def log_ip_guessing_match_metrics(proportion)
@importer_stats.gauge('ip_proportion', proportion)
end
def is_ip_column?(column)
return false unless is_text_type? column
proportion = ip_proportion(column)
if proportion > threshold
log "ip_proportion(#{column[:column_name]}) = #{proportion}; threshold = #{threshold}; sample.count = #{sample.count}"
log "sample.first(4) = #{sample.first(4)}"
log_ip_guessing_match_metrics(proportion)
true
else
false
end
end
# See http://en.wikipedia.org/wiki/Entropy_(information_theory)
# See http://www.shannonentropy.netmark.pl/
#
# Returns 0.0 if all elements in the column are repeated
# Returns 1.0 if all elements in the column are different
def metric_entropy(column, normalizer=nil)
shannon_entropy(column, normalizer) / Math.log(sample.count)
end
def shannon_entropy(column, normalizer)
sum = 0.0
frequencies(column, normalizer).each { |freq| sum += (freq * Math.log(freq)) }
return sum.abs
end
# Returns an array with the relative frequencies of the elements of that column
def frequencies(column, normalizer)
frequency_table = {}
column_name_sym = column[:column_name].to_sym
if normalizer
sample.each do |row|
elem = normalizer.call(row[column_name_sym])
update_frequency_element(frequency_table, elem)
end
else
sample.each do |row|
elem = row[column_name_sym]
update_frequency_element(frequency_table, elem)
end
end
length = sample.count.to_f
frequency_table.map { |key, value| value / length }
end
def country_proportion(column)
column_name_sym = column[:column_name].to_sym
matches = sample.count { |row| countries.include? country_name_normalizer.call(row[column_name_sym]) }
country_proportion = matches.to_f / sample.count
log "country_proportion(#{column[:column_name]}) = #{country_proportion}"
country_proportion
end
def log(msg)
@job.log msg if @job
end
def sample
@sample ||= TableSampler.new(@db, qualified_table_name, id_column, sample_size).sample
end
def id_column
return @id_column if @id_column
columns.each do |column|
if ID_COLUMNS.include? column[:column_name]
@id_column = column[:column_name]
return @id_column
end
end
raise ContentGuesserException, "Couldn't find an id column for table #{qualified_table_name}"
end
def ip_proportion(column)
column_name_sym = column[:column_name].to_sym
matches = sample.count { |row| IpChecker.is_ip?(row[column_name_sym]) }
matches.to_f / sample.count
end
def threshold
@options[:guessing][:threshold]
end
def is_text_type? column
['character varying', 'varchar', 'text'].include? column[:data_type]
end
def sample_size
@options[:guessing][:sample_size]
end
def minimum_entropy
@minimum_entropy ||= @options[:guessing].fetch(:minimum_entropy, DEFAULT_MINIMUM_ENTROPY)
end
def countries
return @countries if @countries
@countries = Set.new()
geocoder_sql_api.fetch(COUNTRIES_QUERY).each do |country|
country_name = country[COUNTRIES_COLUMN]
@countries.add country_name if country_name.length >= 2
end
@countries
end
def geocoder_sql_api
@geocoder_sql_api ||= CartoDB::SQLApi.new(
@options[:geocoder][:internal].merge({ timeout: SQLAPI_CALLS_TIMEOUT })
)
end
attr_writer :geocoder_sql_api
def qualified_table_name
%Q("#{@schema}"."#{@table_name}")
end
private
def update_frequency_element(frequency_table, elem)
if frequency_table.key?(elem)
frequency_table[elem] += 1
else
frequency_table[elem] = 1
end
end
end
class ContentGuesserException < StandardError; end
end
end
@@ -0,0 +1,272 @@
require 'csv'
require 'charlock_holmes'
require 'tempfile'
require 'fileutils'
require_relative './job'
require_relative './source_file'
require_relative './unp'
module CartoDB
module Importer2
class CsvNormalizer
LINE_SIZE_FOR_CLEANING = 5000
LINES_FOR_DETECTION = 100 # How many lines to read?
SAMPLE_READ_LIMIT = 500000 # Read big enough sample bytes for the encoding sampling
COMMON_DELIMITERS = [',', "\t", ' ', ';', '|'].freeze
DELIMITER_WEIGHTS = { ',' => 2, "\t" => 2, ' ' => 1, ';' => 2, '|' => 2 }.freeze
DEFAULT_DELIMITER = ','
DEFAULT_ENCODING = 'UTF-8'
DEFAULT_QUOTE = '"'
OUTPUT_DELIMITER = ',' # Normalized CSVs will use this delimiter
ENCODING_CONFIDENCE = 28
ACCEPTABLE_ENCODINGS = %w{ ISO-8859-1 ISO-8859-2 UTF-8 }
REVERSE_LINE_FEED = "\x8D"
def initialize(filepath, job = nil, importer_config = nil)
@filepath = filepath
@job = job || Job.new
@delimiter = nil
@force_normalize = false
@encoding = nil
@importer_config = importer_config
end
def force_normalize
@force_normalize = true
end
# @throws MalformedCSVException
def run
return self unless File.exists?(filepath)
detect_delimiter
begin
return self unless (needs_normalization? || @force_normalize)
rescue CSV::MalformedCSVError => ex
raise MalformedCSVException.new(ex.message)
end
normalize(temporary_filepath)
release
File.rename(temporary_filepath, filepath)
FileUtils.rm_rf(temporary_directory)
self.temporary_directory = nil
self
end
def detect_delimiter
# Calculate variances of the N first lines for each delimiter, then grab the one that changes less
@delimiter = DEFAULT_DELIMITER unless first_line
lines_for_detection = Array.new
LINES_FOR_DETECTION.times {
line = stream.gets
lines_for_detection << remove_quoted_strings(line) unless line.nil?
}
stream.rewind
# Maybe gets was not able to discern line breaks, try manually:
if lines_for_detection.size == 1
lines_for_detection = lines_for_detection.first
# Did it read as columns instead of rows?
if lines_for_detection.class == Array
lines_for_detection.first
end
# Carriage return without newline
lines_for_detection = lines_for_detection.split("\x0D")
end
occurrences = Hash[
COMMON_DELIMITERS.map { |delimiter|
[delimiter, lines_for_detection.map { |line|
line.count(delimiter) }]
}
]
stream.rewind
variances = Hash.new
@delimiter = DEFAULT_DELIMITER
use_variance = true
occurrences.each { |key, values|
if values.length > 1
variances[key] = sample_variance(values) unless values.first == 0
elsif values.length == 1
# If only detected a single line of data, cannot use variance
variances[key] = values.first * DELIMITER_WEIGHTS[key]
use_variance = false
else
use_variance = false
end
}
if variances.length > 0
if use_variance
@delimiter = variances.sort {|a, b| a.last <=> b.last }.first.first
else
# Use whatever delimiter appears more and hope for the best
@delimiter = variances.sort {|a, b| b.last <=> a.last }.first.first
end
end
@delimiter
end
def self.supported?(extension)
%w(.csv .tsv .txt).include?(extension)
end
def normalize(temporary_filepath)
temporary_csv = CSV.open(temporary_filepath, 'w', col_sep: OUTPUT_DELIMITER, encoding: 'UTF-8')
CSV.open(filepath, "rb:#{encoding}", col_sep: @delimiter) do |input|
loop do
begin
row = input.shift
break unless row
rescue CSV::MalformedCSVError
next
end
temporary_csv << multiple_column(row)
end
end
# TODO: it would be nice to detect and warn the user about ignored
# malformed rows (but probably not about malformed empty lines, such
# as trailing \n\r\n seen in some cases)
temporary_csv.close
@delimiter = OUTPUT_DELIMITER
rescue ArgumentError, Encoding::UndefinedConversionError, Encoding::InvalidByteSequenceError => e
raise EncodingDetectionError
end
def temporary_filepath(filename_prefix = '')
File.join(temporary_directory, filename_prefix + File.basename(filepath))
end
def csv_options
{
col_sep: delimiter,
quote_char: DEFAULT_QUOTE
}
end
def needs_normalization?
(!ACCEPTABLE_ENCODINGS.include?(encoding)) ||
(delimiter != DEFAULT_DELIMITER) ||
single_column?
end
def single_column?
columns = ::CSV.parse(first_line, csv_options)
raise EmptyFileError.new if !columns.any?
columns.first.length < 2
end
def multiple_column(row)
return row if row.length > 1
row << nil
end
def delimiter
@delimiter
end
def encoding
return @encoding unless @encoding.nil?
source_file = SourceFile.new(filepath)
if source_file.encoding
@encoding = source_file.encoding
else
data = File.open(filepath, 'r')
sample = data.read(SAMPLE_READ_LIMIT)
data.close
result = CharlockHolmes::EncodingDetector.detect(sample)
# Looks like an ICU problem https://github.com/brianmario/charlock_holmes/issues/38
@encoding = if result.fetch(:encoding, 'UTF-8') == 'IBM424_rtl'
DEFAULT_ENCODING
elsif result.fetch(:confidence, 0) < ENCODING_CONFIDENCE
DEFAULT_ENCODING
else
result.fetch(:encoding, DEFAULT_ENCODING)
end
end
@encoding
rescue
DEFAULT_ENCODING
end
def first_line
return @first_line if @first_line
stream.rewind
@first_line ||= stream.gets || ''
stream.rewind
@first_line
end
def release
@stream.close
@stream = nil
@first_line = nil
self
end
def stream
@stream ||= File.open(filepath, 'rb')
end
attr_reader :filepath
alias_method :converted_filepath, :filepath
private
def generate_temporary_directory
self.temporary_directory = Unp.new(@importer_config).generate_temporary_directory.temporary_directory
self
end
def temporary_directory
generate_temporary_directory unless @temporary_directory
@temporary_directory
end
def sum(items_list)
items_list.inject(0){|accum, i| accum + i }
end
def mean(items_list)
sum(items_list) / items_list.length.to_f
end
def sample_variance(items_list)
m = mean(items_list)
sum = items_list.inject(0){|accum, i| accum + (i-m)**2 }
sum / (items_list.length - 1).to_f
end
def remove_quoted_strings(input)
# Note that CSV quoted strings can use double quotes, `""`
# as a way of escaping a single quote `"`
# Since we're just removing all quoted strings, this simple
# approach works in that case too.
input.gsub(/"[^\\"]*"/, '')
end
attr_writer :temporary_directory
end
end
end

Some files were not shown because too many files have changed in this diff Show More