Last active
August 29, 2015 14:13
-
-
Save mattyb149/e4cf796ff45983ebf87e to your computer and use it in GitHub Desktop.
Split rows based on a field value in Pentaho Data Integration
This file contains bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
<?xml version="1.0" encoding="UTF-8"?> | |
<transformation> | |
<info> | |
<name>split_field_num_records</name> | |
<description/> | |
<extended_description/> | |
<trans_version/> | |
<trans_type>Normal</trans_type> | |
<directory>/</directory> | |
<parameters> | |
</parameters> | |
<log> | |
<trans-log-table><connection/> | |
<schema/> | |
<table/> | |
<size_limit_lines/> | |
<interval/> | |
<timeout_days/> | |
<field><id>ID_BATCH</id><enabled>Y</enabled><name>ID_BATCH</name></field><field><id>CHANNEL_ID</id><enabled>Y</enabled><name>CHANNEL_ID</name></field><field><id>TRANSNAME</id><enabled>Y</enabled><name>TRANSNAME</name></field><field><id>STATUS</id><enabled>Y</enabled><name>STATUS</name></field><field><id>LINES_READ</id><enabled>Y</enabled><name>LINES_READ</name><subject/></field><field><id>LINES_WRITTEN</id><enabled>Y</enabled><name>LINES_WRITTEN</name><subject/></field><field><id>LINES_UPDATED</id><enabled>Y</enabled><name>LINES_UPDATED</name><subject/></field><field><id>LINES_INPUT</id><enabled>Y</enabled><name>LINES_INPUT</name><subject/></field><field><id>LINES_OUTPUT</id><enabled>Y</enabled><name>LINES_OUTPUT</name><subject/></field><field><id>LINES_REJECTED</id><enabled>Y</enabled><name>LINES_REJECTED</name><subject/></field><field><id>ERRORS</id><enabled>Y</enabled><name>ERRORS</name></field><field><id>STARTDATE</id><enabled>Y</enabled><name>STARTDATE</name></field><field><id>ENDDATE</id><enabled>Y</enabled><name>ENDDATE</name></field><field><id>LOGDATE</id><enabled>Y</enabled><name>LOGDATE</name></field><field><id>DEPDATE</id><enabled>Y</enabled><name>DEPDATE</name></field><field><id>REPLAYDATE</id><enabled>Y</enabled><name>REPLAYDATE</name></field><field><id>LOG_FIELD</id><enabled>Y</enabled><name>LOG_FIELD</name></field><field><id>EXECUTING_SERVER</id><enabled>N</enabled><name>EXECUTING_SERVER</name></field><field><id>EXECUTING_USER</id><enabled>N</enabled><name>EXECUTING_USER</name></field><field><id>CLIENT</id><enabled>N</enabled><name>CLIENT</name></field></trans-log-table> | |
<perf-log-table><connection/> | |
<schema/> | |
<table/> | |
<interval/> | |
<timeout_days/> | |
<field><id>ID_BATCH</id><enabled>Y</enabled><name>ID_BATCH</name></field><field><id>SEQ_NR</id><enabled>Y</enabled><name>SEQ_NR</name></field><field><id>LOGDATE</id><enabled>Y</enabled><name>LOGDATE</name></field><field><id>TRANSNAME</id><enabled>Y</enabled><name>TRANSNAME</name></field><field><id>STEPNAME</id><enabled>Y</enabled><name>STEPNAME</name></field><field><id>STEP_COPY</id><enabled>Y</enabled><name>STEP_COPY</name></field><field><id>LINES_READ</id><enabled>Y</enabled><name>LINES_READ</name></field><field><id>LINES_WRITTEN</id><enabled>Y</enabled><name>LINES_WRITTEN</name></field><field><id>LINES_UPDATED</id><enabled>Y</enabled><name>LINES_UPDATED</name></field><field><id>LINES_INPUT</id><enabled>Y</enabled><name>LINES_INPUT</name></field><field><id>LINES_OUTPUT</id><enabled>Y</enabled><name>LINES_OUTPUT</name></field><field><id>LINES_REJECTED</id><enabled>Y</enabled><name>LINES_REJECTED</name></field><field><id>ERRORS</id><enabled>Y</enabled><name>ERRORS</name></field><field><id>INPUT_BUFFER_ROWS</id><enabled>Y</enabled><name>INPUT_BUFFER_ROWS</name></field><field><id>OUTPUT_BUFFER_ROWS</id><enabled>Y</enabled><name>OUTPUT_BUFFER_ROWS</name></field></perf-log-table> | |
<channel-log-table><connection/> | |
<schema/> | |
<table/> | |
<timeout_days/> | |
<field><id>ID_BATCH</id><enabled>Y</enabled><name>ID_BATCH</name></field><field><id>CHANNEL_ID</id><enabled>Y</enabled><name>CHANNEL_ID</name></field><field><id>LOG_DATE</id><enabled>Y</enabled><name>LOG_DATE</name></field><field><id>LOGGING_OBJECT_TYPE</id><enabled>Y</enabled><name>LOGGING_OBJECT_TYPE</name></field><field><id>OBJECT_NAME</id><enabled>Y</enabled><name>OBJECT_NAME</name></field><field><id>OBJECT_COPY</id><enabled>Y</enabled><name>OBJECT_COPY</name></field><field><id>REPOSITORY_DIRECTORY</id><enabled>Y</enabled><name>REPOSITORY_DIRECTORY</name></field><field><id>FILENAME</id><enabled>Y</enabled><name>FILENAME</name></field><field><id>OBJECT_ID</id><enabled>Y</enabled><name>OBJECT_ID</name></field><field><id>OBJECT_REVISION</id><enabled>Y</enabled><name>OBJECT_REVISION</name></field><field><id>PARENT_CHANNEL_ID</id><enabled>Y</enabled><name>PARENT_CHANNEL_ID</name></field><field><id>ROOT_CHANNEL_ID</id><enabled>Y</enabled><name>ROOT_CHANNEL_ID</name></field></channel-log-table> | |
<step-log-table><connection/> | |
<schema/> | |
<table/> | |
<timeout_days/> | |
<field><id>ID_BATCH</id><enabled>Y</enabled><name>ID_BATCH</name></field><field><id>CHANNEL_ID</id><enabled>Y</enabled><name>CHANNEL_ID</name></field><field><id>LOG_DATE</id><enabled>Y</enabled><name>LOG_DATE</name></field><field><id>TRANSNAME</id><enabled>Y</enabled><name>TRANSNAME</name></field><field><id>STEPNAME</id><enabled>Y</enabled><name>STEPNAME</name></field><field><id>STEP_COPY</id><enabled>Y</enabled><name>STEP_COPY</name></field><field><id>LINES_READ</id><enabled>Y</enabled><name>LINES_READ</name></field><field><id>LINES_WRITTEN</id><enabled>Y</enabled><name>LINES_WRITTEN</name></field><field><id>LINES_UPDATED</id><enabled>Y</enabled><name>LINES_UPDATED</name></field><field><id>LINES_INPUT</id><enabled>Y</enabled><name>LINES_INPUT</name></field><field><id>LINES_OUTPUT</id><enabled>Y</enabled><name>LINES_OUTPUT</name></field><field><id>LINES_REJECTED</id><enabled>Y</enabled><name>LINES_REJECTED</name></field><field><id>ERRORS</id><enabled>Y</enabled><name>ERRORS</name></field><field><id>LOG_FIELD</id><enabled>N</enabled><name>LOG_FIELD</name></field></step-log-table> | |
<metrics-log-table><connection/> | |
<schema/> | |
<table/> | |
<timeout_days/> | |
<field><id>ID_BATCH</id><enabled>Y</enabled><name>ID_BATCH</name></field><field><id>CHANNEL_ID</id><enabled>Y</enabled><name>CHANNEL_ID</name></field><field><id>LOG_DATE</id><enabled>Y</enabled><name>LOG_DATE</name></field><field><id>METRICS_DATE</id><enabled>Y</enabled><name>METRICS_DATE</name></field><field><id>METRICS_CODE</id><enabled>Y</enabled><name>METRICS_CODE</name></field><field><id>METRICS_DESCRIPTION</id><enabled>Y</enabled><name>METRICS_DESCRIPTION</name></field><field><id>METRICS_SUBJECT</id><enabled>Y</enabled><name>METRICS_SUBJECT</name></field><field><id>METRICS_TYPE</id><enabled>Y</enabled><name>METRICS_TYPE</name></field><field><id>METRICS_VALUE</id><enabled>Y</enabled><name>METRICS_VALUE</name></field></metrics-log-table> | |
</log> | |
<maxdate> | |
<connection/> | |
<table/> | |
<field/> | |
<offset>0.0</offset> | |
<maxdiff>0.0</maxdiff> | |
</maxdate> | |
<size_rowset>10000</size_rowset> | |
<sleep_time_empty>50</sleep_time_empty> | |
<sleep_time_full>50</sleep_time_full> | |
<unique_connections>N</unique_connections> | |
<feedback_shown>Y</feedback_shown> | |
<feedback_size>50000</feedback_size> | |
<using_thread_priorities>Y</using_thread_priorities> | |
<shared_objects_file/> | |
<capture_step_performance>N</capture_step_performance> | |
<step_performance_capturing_delay>1000</step_performance_capturing_delay> | |
<step_performance_capturing_size_limit>100</step_performance_capturing_size_limit> | |
<dependencies> | |
</dependencies> | |
<partitionschemas> | |
</partitionschemas> | |
<slaveservers> | |
</slaveservers> | |
<clusterschemas> | |
</clusterschemas> | |
<created_user>-</created_user> | |
<created_date>2015/01/08 10:05:41.537</created_date> | |
<modified_user>-</modified_user> | |
<modified_date>2015/01/08 10:05:41.537</modified_date> | |
</info> | |
<notepads> | |
</notepads> | |
<order> | |
<hop> <from>Calc seat #</from><to>Select columns</to><enabled>Y</enabled> </hop> | |
<hop> <from>Data Grid</from><to>Clone seats</to><enabled>Y</enabled> </hop> | |
<hop> <from>Clone seats</from><to>Exclude original row</to><enabled>Y</enabled> </hop> | |
<hop> <from>Exclude original row</from><to>Calc seat #</to><enabled>Y</enabled> </hop> | |
</order> | |
<step> | |
<name>Data Grid</name> | |
<type>DataGrid</type> | |
<description/> | |
<distribute>Y</distribute> | |
<custom_distribution/> | |
<copies>1</copies> | |
<partitioning> | |
<method>none</method> | |
<schema_name/> | |
</partitioning> | |
<fields> | |
<field> | |
<name>event_name</name> | |
<type>String</type> | |
<format/> | |
<currency/> | |
<decimal/> | |
<group/> | |
<length>-1</length> | |
<precision>-1</precision> | |
<set_empty_string>N</set_empty_string> | |
</field> | |
<field> | |
<name>section_name</name> | |
<type>String</type> | |
<format/> | |
<currency/> | |
<decimal/> | |
<group/> | |
<length>-1</length> | |
<precision>-1</precision> | |
<set_empty_string>N</set_empty_string> | |
</field> | |
<field> | |
<name>row_name</name> | |
<type>String</type> | |
<format/> | |
<currency/> | |
<decimal/> | |
<group/> | |
<length>-1</length> | |
<precision>-1</precision> | |
<set_empty_string>N</set_empty_string> | |
</field> | |
<field> | |
<name>num_seats</name> | |
<type>Integer</type> | |
<format/> | |
<currency/> | |
<decimal/> | |
<group/> | |
<length>-1</length> | |
<precision>-1</precision> | |
<set_empty_string>N</set_empty_string> | |
</field> | |
<field> | |
<name>seat_num</name> | |
<type>Integer</type> | |
<format/> | |
<currency/> | |
<decimal/> | |
<group/> | |
<length>-1</length> | |
<precision>-1</precision> | |
<set_empty_string>N</set_empty_string> | |
</field> | |
<field> | |
<name>last_seat</name> | |
<type>Integer</type> | |
<format/> | |
<currency/> | |
<decimal/> | |
<group/> | |
<length>-1</length> | |
<precision>-1</precision> | |
<set_empty_string>N</set_empty_string> | |
</field> | |
<field> | |
<name>client_name</name> | |
<type>String</type> | |
<format/> | |
<currency/> | |
<decimal/> | |
<group/> | |
<length>-1</length> | |
<precision>-1</precision> | |
<set_empty_string>N</set_empty_string> | |
</field> | |
<field> | |
<name>venue_name</name> | |
<type>String</type> | |
<format/> | |
<currency/> | |
<decimal/> | |
<group/> | |
<length>-1</length> | |
<precision>-1</precision> | |
<set_empty_string>N</set_empty_string> | |
</field> | |
<field> | |
<name>file_prefix</name> | |
<type>String</type> | |
<format/> | |
<currency/> | |
<decimal/> | |
<group/> | |
<length>-1</length> | |
<precision>-1</precision> | |
<set_empty_string>N</set_empty_string> | |
</field> | |
</fields> | |
<data> | |
<line> <item>SJQ0802</item><item>122</item><item>28</item><item>4</item><item>6</item><item>9</item><item>49ers</item><item>Levis Stadium</item><item>sbl49ers</item> </line> | |
</data> | |
<cluster_schema/> | |
<remotesteps> <input> </input> <output> </output> </remotesteps> <GUI> | |
<xloc>34</xloc> | |
<yloc>32</yloc> | |
<draw>Y</draw> | |
</GUI> | |
</step> | |
<step> | |
<name>Calc seat #</name> | |
<type>Calculator</type> | |
<description/> | |
<distribute>Y</distribute> | |
<custom_distribution/> | |
<copies>1</copies> | |
<partitioning> | |
<method>none</method> | |
<schema_name/> | |
</partitioning> | |
<calculation><field_name>ONE</field_name> | |
<calc_type>CONSTANT</calc_type> | |
<field_a>1</field_a> | |
<field_b/> | |
<field_c/> | |
<value_type>Integer</value_type> | |
<value_length>-1</value_length> | |
<value_precision>-1</value_precision> | |
<remove>N</remove> | |
<conversion_mask/> | |
<decimal_symbol/> | |
<grouping_symbol/> | |
<currency_symbol/> | |
</calculation> | |
<calculation><field_name>temp</field_name> | |
<calc_type>ADD</calc_type> | |
<field_a>seat_num</field_a> | |
<field_b>seat_index_rownum</field_b> | |
<field_c/> | |
<value_type>Integer</value_type> | |
<value_length>-1</value_length> | |
<value_precision>-1</value_precision> | |
<remove>Y</remove> | |
<conversion_mask/> | |
<decimal_symbol/> | |
<grouping_symbol/> | |
<currency_symbol/> | |
</calculation> | |
<calculation><field_name>seatNum</field_name> | |
<calc_type>SUBTRACT</calc_type> | |
<field_a>temp</field_a> | |
<field_b>ONE</field_b> | |
<field_c/> | |
<value_type>Integer</value_type> | |
<value_length>-1</value_length> | |
<value_precision>-1</value_precision> | |
<remove>N</remove> | |
<conversion_mask/> | |
<decimal_symbol/> | |
<grouping_symbol/> | |
<currency_symbol/> | |
</calculation> | |
<cluster_schema/> | |
<remotesteps> <input> </input> <output> </output> </remotesteps> <GUI> | |
<xloc>388</xloc> | |
<yloc>32</yloc> | |
<draw>Y</draw> | |
</GUI> | |
</step> | |
<step> | |
<name>Select columns</name> | |
<type>SelectValues</type> | |
<description/> | |
<distribute>Y</distribute> | |
<custom_distribution/> | |
<copies>1</copies> | |
<partitioning> | |
<method>none</method> | |
<schema_name/> | |
</partitioning> | |
<fields> <field> <name>event_name</name> | |
<rename/> | |
<length>-2</length> | |
<precision>-2</precision> | |
</field> <field> <name>section_name</name> | |
<rename/> | |
<length>-2</length> | |
<precision>-2</precision> | |
</field> <field> <name>row_name</name> | |
<rename/> | |
<length>-2</length> | |
<precision>-2</precision> | |
</field> <field> <name>ONE</name> | |
<rename>num_seats</rename> | |
<length>-2</length> | |
<precision>-2</precision> | |
</field> <field> <name>seatNum</name> | |
<rename>seat_num</rename> | |
<length>-2</length> | |
<precision>-2</precision> | |
</field> <field> <name>seatNum</name> | |
<rename>last_seat</rename> | |
<length>-2</length> | |
<precision>-2</precision> | |
</field> <field> <name>client_name</name> | |
<rename/> | |
<length>-2</length> | |
<precision>-2</precision> | |
</field> <field> <name>venue_name</name> | |
<rename/> | |
<length>-2</length> | |
<precision>-2</precision> | |
</field> <field> <name>file_prefix</name> | |
<rename/> | |
<length>-2</length> | |
<precision>-2</precision> | |
</field> <select_unspecified>N</select_unspecified> | |
</fields> <cluster_schema/> | |
<remotesteps> <input> </input> <output> </output> </remotesteps> <GUI> | |
<xloc>497</xloc> | |
<yloc>32</yloc> | |
<draw>Y</draw> | |
</GUI> | |
</step> | |
<step> | |
<name>Clone seats</name> | |
<type>CloneRow</type> | |
<description/> | |
<distribute>Y</distribute> | |
<custom_distribution/> | |
<copies>1</copies> | |
<partitioning> | |
<method>none</method> | |
<schema_name/> | |
</partitioning> | |
<nrclones>0</nrclones> | |
<addcloneflag>Y</addcloneflag> | |
<cloneflagfield>clone?</cloneflagfield> | |
<nrcloneinfield>Y</nrcloneinfield> | |
<nrclonefield>num_seats</nrclonefield> | |
<addclonenum>Y</addclonenum> | |
<clonenumfield>seat_index_rownum</clonenumfield> | |
<cluster_schema/> | |
<remotesteps> <input> </input> <output> </output> </remotesteps> <GUI> | |
<xloc>136</xloc> | |
<yloc>32</yloc> | |
<draw>Y</draw> | |
</GUI> | |
</step> | |
<step> | |
<name>Exclude original row</name> | |
<type>FilterRows</type> | |
<description/> | |
<distribute>Y</distribute> | |
<custom_distribution/> | |
<copies>1</copies> | |
<partitioning> | |
<method>none</method> | |
<schema_name/> | |
</partitioning> | |
<send_true_to>Calc seat #</send_true_to> | |
<send_false_to/> | |
<compare> | |
<condition> | |
<negated>N</negated> | |
<leftvalue>clone?</leftvalue> | |
<function>=</function> | |
<rightvalue/> | |
<value><name>constant</name><type>Boolean</type><text>Y</text><length>-1</length><precision>-1</precision><isnull>N</isnull><mask/></value> </condition> | |
</compare> | |
<cluster_schema/> | |
<remotesteps> <input> </input> <output> </output> </remotesteps> <GUI> | |
<xloc>260</xloc> | |
<yloc>32</yloc> | |
<draw>Y</draw> | |
</GUI> | |
</step> | |
<step_error_handling> | |
</step_error_handling> | |
<slave-step-copy-partition-distribution> | |
</slave-step-copy-partition-distribution> | |
<slave_transformation>N</slave_transformation> | |
</transformation> |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment