<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Airflow on p &lt; Whatever</title>
    <link>https://szeitlin.github.io/posts/airflow/</link>
    <description>Recent content in Airflow on p &lt; Whatever</description>
    <generator>Hugo</generator>
    <language>en-us</language>
    <lastBuildDate>Sat, 18 Nov 2017 00:00:00 +0000</lastBuildDate>
    <atom:link href="https://szeitlin.github.io/posts/airflow/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>Airflow</title>
      <link>https://szeitlin.github.io/posts/airflow/airflow-for-hands-off-etl/</link>
      <pubDate>Sat, 18 Nov 2017 00:00:00 +0000</pubDate>
      <guid>https://szeitlin.github.io/posts/airflow/airflow-for-hands-off-etl/</guid>
      <description>&lt;h1 id=&#34;airflow-for-hands-off-etl&#34;&gt;Airflow for hands-off ETL&lt;/h1&gt;&#xA;&lt;hr&gt;&#xA;&lt;p&gt;Almost exactly a year ago, I joined &lt;a href=&#34;https://www.yahoo.com&#34;&gt;Yahoo&lt;/a&gt;, which more recently became &lt;a href=&#34;https://www.oath.com&#34;&gt;Oath&lt;/a&gt;.&lt;/p&gt;&#xA;&lt;p&gt;The team I joined is called the Product Hackers, and we work with large amounts of data. By large amounts I meant, billions of rows of log data.&lt;/p&gt;&#xA;&lt;p&gt;Our team does both ad-hoc analyses and ongoing machine learning projects. In order to support those efforts, our team had initially written scripts to parse logs and run them with cron to load the data into Redshift on AWS. After a while, it made sense to move to &lt;a href=&#34;http://pythonhosted.org/airflow/&#34;&gt;Airflow&lt;/a&gt;.&lt;/p&gt;</description>
    </item>
  </channel>
</rss>
