diff --git a/docs/models/Batch Presentation Diagrams-1.vsd b/docs/models/Batch Presentation Diagrams-1.vsd new file mode 100644 index 000000000..0c589145d Binary files /dev/null and b/docs/models/Batch Presentation Diagrams-1.vsd differ diff --git a/docs/models/Batch Presentation Diagrams.vsd b/docs/models/Batch Presentation Diagrams.vsd new file mode 100644 index 000000000..d25c7da09 Binary files /dev/null and b/docs/models/Batch Presentation Diagrams.vsd differ diff --git a/docs/models/Figures.ppt b/docs/models/Figures.ppt new file mode 100644 index 000000000..8ad87a7c1 Binary files /dev/null and b/docs/models/Figures.ppt differ diff --git a/docs/models/batch-architecture-review.doc b/docs/models/batch-architecture-review.doc new file mode 100644 index 000000000..50a7eaaa8 Binary files /dev/null and b/docs/models/batch-architecture-review.doc differ diff --git a/docs/ppt/Batch Presentation Diagrams-1.vsd b/docs/ppt/Batch Presentation Diagrams-1.vsd new file mode 100644 index 000000000..0c589145d Binary files /dev/null and b/docs/ppt/Batch Presentation Diagrams-1.vsd differ diff --git a/docs/ppt/Batch Presentation Diagrams.vsd b/docs/ppt/Batch Presentation Diagrams.vsd new file mode 100644 index 000000000..d25c7da09 Binary files /dev/null and b/docs/ppt/Batch Presentation Diagrams.vsd differ diff --git a/docs/ppt/Figures.ppt b/docs/ppt/Figures.ppt new file mode 100644 index 000000000..8ad87a7c1 Binary files /dev/null and b/docs/ppt/Figures.ppt differ diff --git a/docs/src/docbkx/resources/css/html.css b/docs/src/docbkx/resources/css/html.css new file mode 100644 index 000000000..36297ac36 --- /dev/null +++ b/docs/src/docbkx/resources/css/html.css @@ -0,0 +1,421 @@ +body { + text-align: justify; + margin-right: 2em; + margin-left: 2em; +} + +a, + a[accesskey^ + += +"h" +] +, +a[accesskey^ + += +"n" +] +, +a[accesskey^ + += +"u" +] +, +a[accesskey^ + += +"p" +] +{ +font-family: Verdana, Arial, helvetica, sans-serif + +; +font-size: + +12 +px + +; +color: #003399 + +; +} + +a:active { + color: #003399; +} + +a:visited { + color: #888888; +} + +p { + font-family: Verdana, Arial, sans-serif; +} + +dt { + font-family: Verdana, Arial, sans-serif; + font-size: 12px; +} + +p, dl, dt, dd, blockquote { + color: #000000; + margin-bottom: 3px; + margin-top: 3px; + padding-top: 0px; +} + +ol, ul, p { + margin-top: 6px; + margin-bottom: 6px; +} + +p, blockquote { + font-size: 90%; +} + +p.releaseinfo { + font-size: 100%; + font-weight: bold; + font-family: Verdana, Arial, helvetica, sans-serif; + padding-top: 10px; +} + +p.pubdate { + font-size: 120%; + font-weight: bold; + font-family: Verdana, Arial, helvetica, sans-serif; +} + +td { + font-size: 80%; +} + +td, th, span { + color: #000000; +} + +td[width^ + += +"40%" +] +{ +font-family: Verdana, Arial, helvetica, sans-serif + +; +font-size: + +12 +px + +; +color: #003399 + +; +} + +table[summary^ + += +"Navigation header" +] +tbody tr th[colspan^ + += +"3" +] +{ +font-family: Verdana, Arial, helvetica, sans-serif + +; +} + +blockquote { + margin-right: 0px; +} + +h1, h2, h3, h4, h6, H6 { + color: #000000; + font-weight: 500; + margin-top: 0px; + padding-top: 14px; + font-family: Verdana, Arial, helvetica, sans-serif; + margin-bottom: 0px; +} + +h2.title { + font-weight: 800; + margin-bottom: 8px; +} + +h2.subtitle { + font-weight: 800; + margin-bottom: 20px; +} + +.firstname, .surname { + font-size: 12px; + font-family: Verdana, Arial, helvetica, sans-serif; +} + +table { + border-collapse: collapse; + border-spacing: 0; + border: 1px black; + empty-cells: hide; + margin: 10px 0px 30px 50px; + width: 90%; +} + +div.table { + margin: 30px 0px 30px 0px; + border: 1px dashed gray; + padding: 10px; +} + +div .table-contents table { + border: 1px solid black; +} + +div.table > p.title { + padding-left: 10px; +} + +table[summary^ + += +"Navigation footer" +] +{ +border-collapse: collapse + +; +border-spacing: + +0 +; +border: + +1 +px black + +; +empty-cells: hide + +; +margin: + +0 +px + +; +width: + +100 +% +; +} + +table[summary^ + += +"Note" +] +, +table[summary^ + += +"Warning" +] +, +table[summary^ + += +"Tip" +] +{ +border-collapse: collapse + +; +border-spacing: + +0 +; +border: + +1 +px black + +; +empty-cells: hide + +; +margin: + +10 +px + +0 +px + +10 +px + +- +20 +px + +; +width: + +100 +% +; +} + +td { + padding: 4pt; + font-family: Verdana, Arial, helvetica, sans-serif; +} + +div.warning TD { + text-align: justify; +} + +h1 { + font-size: 150%; +} + +h2 { + font-size: 110%; +} + +h3 { + font-size: 100%; + font-weight: bold; +} + +h4 { + font-size: 90%; + font-weight: bold; +} + +h5 { + font-size: 90%; + font-style: italic; +} + +h6 { + font-size: 100%; + font-style: italic; +} + +tt { + font-size: 110%; + font-family: "Courier New", Courier, monospace; + color: #000000; +} + +.navheader, .navfooter { + border: none; +} + +div.navfooter table { + border: dashed gray; + border-width: 1px 1px 1px 1px; + background-color: #cde48d; +} + +pre { + font-size: 110%; + padding: 5px; + border-style: solid; + border-width: 1px; + border-color: #CCCCCC; + background-color: #f3f5e9; +} + +ul, ol, li { + list-style: disc; +} + +hr { + width: 100%; + height: 1px; + background-color: #CCCCCC; + border-width: 0px; + padding: 0px; +} + +.variablelist { + padding-top: 10px; + padding-bottom: 10px; + margin: 0; +} + +.term { + font-weight: bold; +} + +.mediaobject { + padding-top: 30px; + padding-bottom: 30px; +} + +.legalnotice { + font-family: Verdana, Arial, helvetica, sans-serif; + font-size: 12px; + font-style: italic; +} + +.sidebar { + float: right; + margin: 10px 0px 10px 30px; + padding: 10px 20px 20px 20px; + width: 33%; + border: 1px solid black; + background-color: #F4F4F4; + font-size: 14px; +} + +.property { + font-family: "Courier New", Courier, monospace; +} + +a code { + font-family: Verdana, Arial, monospace; + font-size: 12px; +} + +td code { + font-size: 110%; +} + +div.note * td, + div.tip * td, + div.warning * td, + div.calloutlist * td { + text-align: justify; + font-size: 100%; +} + +.programlisting .interfacename, + .programlisting .literal, + .programlisting .classname { + font-size: 95%; +} + +.title .interfacename, + .title .literal, + .title .classname { + font-size: 130%; +} + +/* everything in a is displayed in a coloured, comment-like font */ +.programlisting * .lineannotation, + .programlisting * .lineannotation * { + color: green; +} diff --git a/docs/src/docbkx/resources/images/i21-banner-rhs.jpg b/docs/src/docbkx/resources/images/i21-banner-rhs.jpg new file mode 100644 index 000000000..8b24a7736 Binary files /dev/null and b/docs/src/docbkx/resources/images/i21-banner-rhs.jpg differ diff --git a/docs/src/docbkx/resources/images/xdev-spring_logo.jpg b/docs/src/docbkx/resources/images/xdev-spring_logo.jpg new file mode 100644 index 000000000..622962ee3 Binary files /dev/null and b/docs/src/docbkx/resources/images/xdev-spring_logo.jpg differ diff --git a/docs/src/docbkx/resources/xsl/fopdf.xsl b/docs/src/docbkx/resources/xsl/fopdf.xsl new file mode 100644 index 000000000..b51a94032 --- /dev/null +++ b/docs/src/docbkx/resources/xsl/fopdf.xsl @@ -0,0 +1,468 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + Copyright © 2005-2007 + + + , + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + -5em + -5em + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + bold + + + + + + + + + + + + + + + + + + + + + + + + + + + 1 + 0 + 1 + + 1 + + + + + + book toc + + + + 2 + + + + + + + + + + 0 + 0 + 0 + + + 5mm + 10mm + 10mm + + 15mm + 10mm + 0mm + + 18mm + 18mm + + + 0pc + + + + + justify + false + + + 11 + 8 + + + 1.4 + + + + + + + 0.8em + + + + + + 17.4cm + + + + 4pt + 4pt + 4pt + 4pt + + + + 0.1pt + 0.1pt + + + + + 1 + + + + + + + + left + bold + + + pt + + + + + + + + + + + + + + + 0.8em + 0.8em + 0.8em + + + pt + + 0.1em + 0.1em + 0.1em + + + 0.6em + 0.6em + 0.6em + + + pt + + 0.1em + 0.1em + 0.1em + + + 0.4em + 0.4em + 0.4em + + + pt + + 0.1em + 0.1em + 0.1em + + + + + bold + + + pt + + false + 0.4em + 0.6em + 0.8em + + + + + + + + + pt + + + + + 1em + 1em + 1em + #444444 + solid + 0.1pt + 0.5em + 0.5em + 0.5em + 0.5em + 0.5em + 0.5em + + + + 1 + + #F0F0F0 + + + + + + 0 + 1 + + + 90 + + + + + '1' + + + + + + + figure after + example before + equation before + table before + procedure before + + + + 1 + + + + 0.8em + 0.8em + 0.8em + 0.1em + 0.1em + 0.1em + + + + + + + + + + + + + + + + + diff --git a/docs/src/docbkx/resources/xsl/html.xsl b/docs/src/docbkx/resources/xsl/html.xsl new file mode 100644 index 000000000..aa7930bab --- /dev/null +++ b/docs/src/docbkx/resources/xsl/html.xsl @@ -0,0 +1,91 @@ + + + + + + + + + html.css + + + 1 + 0 + 1 + 0 + + + + + + book toc + + + + 3 + + + + + 1 + + + + + + + 0 + + + 90 + + + + + 0 + + + + + figure after + example before + equation before + table before + procedure before + + + + , + + + + + + + + +
+

Authors

+

+ +

+
+ +
diff --git a/docs/src/docbkx/resources/xsl/html_chunk.xsl b/docs/src/docbkx/resources/xsl/html_chunk.xsl new file mode 100644 index 000000000..232c5d329 --- /dev/null +++ b/docs/src/docbkx/resources/xsl/html_chunk.xsl @@ -0,0 +1,208 @@ + + + + + + + '5' + '1' + html.css + + 1 + 0 + 1 + 0 + + + + book toc + + + 3 + + + 1 + + + + + 90 + + + + figure after + example before + equation before + table before + procedure before + + + + , + + + + + + + + +
+

Authors

+

+ +

+
+ + + + + + + + 1 + + + + + + + + + + + + + +
diff --git a/docs/src/models/flat-file-input-source-design.dnx b/docs/src/models/flat-file-input-source-design.dnx new file mode 100644 index 000000000..0d1bc586f --- /dev/null +++ b/docs/src/models/flat-file-input-source-design.dnx @@ -0,0 +1,283 @@ + + +?> + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/docs/src/models/io-design.dnx b/docs/src/models/io-design.dnx new file mode 100644 index 000000000..57707546f --- /dev/null +++ b/docs/src/models/io-design.dnx @@ -0,0 +1,282 @@ + + +?> + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/docs/src/site/docbook/reference/application.xml b/docs/src/site/docbook/reference/application.xml new file mode 100644 index 000000000..643971432 --- /dev/null +++ b/docs/src/site/docbook/reference/application.xml @@ -0,0 +1,656 @@ + + + + The Nature of Batch Applications + +
+ Batch Application Overview + + The Application layer is anything built on top of the spring batch framework. + Many enterprises solutions are composed of composite applications, meaning online web + applications, SOA Enabled Services, Enterprise Application Integration, and Batch. Despite + the ideal of zero latency applications and the interconnected enterprise many interfaces + still define the exchange of information through flat files. Increasingly these files + are provided in an XML format. The chapter will provide a more in depth coverage of + batch solution space and the kinds of problems that spring batch may provide solutions + for high volume processing. +
+
+ + Batch Processing Strategies + + To help design and implement batch systems, basic batch + application building blocks and patterns should be provided to the + designers and programmers in form of sample structure charts and + code shells. When starting to design a batch job, the business + logic should be decomposed into a series of steps which can be + implemented using the following standard building blocks: + + + + + + Conversion Applications: For each type of file supplied + by or generated to an external system, a conversion application + will need to be created to convert the transaction records + supplied into a standard format required for processing. This + type of batch application can partly or entirely consist of + translation utility modules (see Basic Batch Services). + + + + + + Validation Applications: Validation applications ensure + that all input/output records are correct and consistent. + Validation is typically based on file headers and trailers, + checksums and validation algorithms as well as record level + cross-checks. + + + + + + Extract Applications: An application that reads a set of + records from a database or input file, selects records based on + predefined rules, and writes the records to an output + file. + + + + + + Extract/Update Applications: An application that reads + records from a database or an input file, and makes changes to + a database or an output file driven by the data found in each + input record. + + + + + + Processing and Updating Applications: An application that + performs processing on input transactions from an extract or a + validation application. The processing will usually involve + reading a database to obtain data required for processing, + potentially updating the database and creating records for + output processing. + + + + + + Output/Format Applications: Applications reading an input + file, restructures data from this record according to a + standard format, and produces an output file for printing or + transmission to another program or system. + + + + + + Pre-processing + Capabilities + + Additionally a basic application shell should be provided for + business logic that cannot be built using the previously mentioned + building blocks. + + In addition to the main building blocks, each application may + use one or more of standard utility steps, such as: + + + + + + Sort - A Program that reads an input file and produces an + output file where records have been re-sequenced according to a + sort key field in the records. Sorts are usually performed by + standard system utilities. + + + + + + Split - A program that reads a single input file, and + writes each record to one of several output files based on a + field value. Splits can be tailored or performed by + parameter-driven standard system utilities. + + + + + + Merge - A program that reads records from multiple input + files and produces one output file with combined data from the + input files. Merges can be tailored or performed by + parameter-driven standard system utilities. + + + + + + Batch applications can additionally be categorized by their + input source: + + + + + + Database-driven applications are driven by rows or values + retrieved from the database. + + + + + + File-driven applications are driven by records or values + retrieved from a file + + + + + + The foundation of any batch system is the processing + strategy. Factors affecting the selection of the strategy include + estimated batch system volume, concurrency with on-line or with + another batch systems, available batch windows etc. Also with more + enterprises wanting to be up and running 24x7, it is more challenging to + establish batch windows. + + Typical processing options for batch are: + + + + + + Normal processing in a batch window during off-line + + + + + + Concurrent batch / on-line processing + + + + + + Parallel processing of many different batch runs or jobs + at the same time + + + + + + Streaming i.e. processing of many instances of the same + job at the same time + + + + + + A combination of these + + + + + + The order in the list above reflects the implementation + complexity, processing in a batch window being the easiest and + streaming the most complex to implement. + + Some or all of these options may be supported by a commercial + scheduler. + + In the following section these processing options will be + discussed in more detail. It is important to notice that the commit + and locking strategy adopted by batch processes will be dependent + on the type of processing performed and as a rule of thumb, the + on-line locking should use the same principles. Therefore a batch + architecture cannot be simply an afterthought when designing an + overall architecture. + + The locking strategy can use only normal database locks, or + an additional custom locking service can be implemented in the + architecture. The locking service would track database locking (for + example by storing the necessary information in a dedicated + db-table) and give or deny permissions to the application programs + requesting a db operation. Retry logic could also be implemented by + this architecture to avoid aborting a batch job in case of a lock + situation. + + 1. Normal processing in a batch + window For simple batch processes running in a separate + batch window, where the data being updated is not required by + on-line users or other batch processes, concurrency is not an issue + and a single commit can be done at the end of the batch run. + + In most cases a more robust approach is more appropriate. A + thing to keep in mind is that batch systems have a tendency to grow + as time goes by, both in terms of complexity and the data volumes + they will handle. If no locking strategy is in place and the system + still relies on a single commit point, modifying the batch programs + can be painful. Therefore, even with the simplest batch systems, + consider the need for commit logic depicted in the + [Restart/Recovery section|Restart & Recovery] as well as the + information concerning the more complex cases below. + + 2. Concurrent batch / on-line + processing Batch applications processing data that can + simultaneously be updated by on-line users, should not lock any + data (either in the database or in files) which could be required + by on-line users for more than a few seconds. Also updates should + be committed to the database at the end of every transaction or what's + referred to as a commit interval of size=1. This minimizes the portion + of data that is unavailable to other processes and the elapsed time the + data is unavailable. + + 2. Concurrent batch / on-line + processing Batch applications processing data that can + simultaneously be updated by on-line users, should not lock any + data (either in the database or in files) which could be required + by on-line users for more than a few seconds. Also updates should + be committed to the database at the end of every transaction or what's + referred to as a commit interval of size=1. This minimizes the portion + of data that is unavailable to other processes and the elapsed time the + data is unavailable. + + Another option to minimize physical locking is to have a + logical row-level locking implemented using either an Optimistic + Locking Pattern or a Pessimistic Locking Pattern. + + + + + + Optimistic locking assumes a low likelihood of record + contention. It typically means inserting a timestamp column in + each database table used concurrently by both batch and on-line + processing. When an application fetches a row for processing, + it also fetches the timestamp. As the application then tries to + update the processed row, the update uses the original + timestamp in the WHERE clause. If the timestamp matches, the + data and the timestamp will be updated successfully. If the + timestamp does not match, this indicates that another + application has updated the same row between the fetch and the + update attempt and therefore the update cannot be + performed. + + + + + + Pessimistic locking is any locking strategy that assumes + there is a high likelihood of record contention and therefore + either a physical or logical lock needs to be obtained at + retrieval time. One type of pessimistic logical locking uses a + dedicated lock-column in the database table. When an + application retrieves the row for update, it sets a flag in the + lock column. With the flag in place, other applications + attempting to retrieve the same row will logically fail. When + the application that set the flag updates the row, it also + clears the flag, enabling the row to be retrieved by other + applications. Please note, that the integrity of data must be + maintained also between the initial fetch and the setting of + the flag, for example by using db locks (e.g.,SELECT FOR + UPDATE). Note also that this method suffers from the same + downside as physical locking except that it is somewhat easier + to manage building a time-out mechanism that will get the lock + released if the user goes to lunch while the record is + locked. + + + + + + These patterns are not necessarily suitable for batch + processing, but they might be used for concurrent batch and on-line + processing for example in cases where the database doesn't + support row-level locking. As a general rule, optimistic locking is + more suitable for on-line applications, while pessimistic locking + is more suitable for batch applications. Whenever logical locking + is used, the same scheme must be used for all applications + accessing data entities protected by logical locks. + + Note that both of these solutions only address locking a + single record. Often we may need to lock a logically related group + of records. With physical locks, you have to manage these very + carefully in order to avoid potential deadlocks. With logical + locks, it is usually best to build a logical lock manager that + understands the logical record groups you want to protect and can + ensure that locks are coherent and non-deadlocking. This logical + lock manager usually uses its own tables for lock management, + contention reporting, time-out mechanism, etc. + + 3. Parallel Processing + Parallel processing allows multiple batch runs / jobs to run in + parallel to minimize the total elapsed batch processing time. This + is not a problem as long as the jobs are not sharing the same + files, db-tables or index spaces. If they do, this service should + be implemented using partitioned data. Another option is to build + an architecture module for maintaining interdependencies using a + control table. A control table should contain a row for each shared + resource and whether it is in use by an application or not. The + batch architecture (Control Program Tasklet) or the application in a + parallel job would then retrieve information from that table to + determine if it can get access to the resource it needs or + not. + + If the data access is not a problem, parallel processing can + be implemented in a mainframe environment using parallel job + classes, in order to ensure adequate CPU time for all the + processes. In an environment other than the mainframe, a similar + solution can be put in place with for example threads. The solution + has to be robust enough to ensure time slices for all the running + processes. + + Other key issues in parallel processing include load + balancing and the availability of general system resources such as + files, database buffer pools etc. Also note that the control table + itself can easily become a critical resource. + + 4. Partitioning Using + partitioning allows multiple versions of large batch applications + to run concurrently. The purpose of this is to reduce the elapsed + time required to process long batch jobs. Processes which can be + successfully partitioned are those where the input file can be + split and/or the main database tables partitioned to allow the + application to run against different sets of data. + + In addition, processes which are partitioned must be designed + to only process their assigned data set. A partitioning + architecture has to be closely tied to the database design and the + database partitioning strategy. Please note, that the database + partitioning doesn't necessarily mean physical partitioning of + the database, although in most cases this is advisable. The + following picture illustrates the partitioning + approach:!app_style_batch_processing.png|align=center! + + The architecture should be flexible enough to allow dynamic + configuration of the number of partitions. Both automatic and user + controlled configuration should be considered. Automatic + configuration may be based on parameters such as the input file + size and/or the number of input records. + + 4.1 Streaming Approaches The + following lists some of the possible streaming approaches. + Selecting a streaming approach has to be done on a case-by-case + basis. + + 1. Fixed and Even Break-Up of Record + Set + + This involves breaking the input record set into an even + number of portions (e.g. 10, where each portion will have exactly + 1/10th of the entire record set). Each portion is then processed by + one instance of the batch/extract application. + + In order to use this approach, preprocessing will be required + to split the recordset up. The result of this split will be a lower + and upper bound placement number which can be used as input to the + batch/extract application in order to restrict its processing to + its portion alone. + + Preprocessing could be a large overhead as it has to + calculate and determine the bounds of each portion of the record + set. + + 2. Breakup by a Key Column + + This involves breaking up the input record set by a key + column such as a location code, and assigning data from each key to + a batch instance. In order to achieve this, column values can + either be + + 3. Assigned to a batch instance via a streaming + table (see below for details). + + 4. Assigned to a batch instance by a portion of the + value (e.g. values 0000-0999, 1000 - 1999, etc.) + + Under option 1, addition of new values will mean a manual + reconfiguration of the batch/extract to ensure that the new value + is added to a particular instance. + + Under option 2, this will ensure that all values are covered + via an instance of the batch job. However, the number of values + processed by one instance is dependent on the distribution of + column values (i.e. there may be a large number of locations in the + 0000-0999 range, and few in the 1000-1999 range). Under this + option, the data range should be designed with streaming in + mind. + + Under both options, the optimal even distribution of records + to batch instances cannot be realized. There is no dynamic + configuration of the number of batch instances used. + + 5. Breakup by Views + + This approach is basically breakup by a key column, but on + the database level. It involves breaking up the recordset into + views. These views will be used by each instance of the batch + application during its processing. The breakup will be done by + grouping the data. + + With this option, each instance of a batch application will + have to be configured to hit a particular view (instead of the + master table). Also, with the addition of new data values, this new + group of data will have to be included into a view. There is no + dynamic configuration capability, as a change in the number of + instances will result in a change to the views. + + 6. Addition of a Processing + Indicator + + This involves the addition of a new column to the input + table, which acts as an indicator. As a preprocessing step, all + indicators would be marked to non-processed. During the record + fetch stage of the batch application, records are read on the + condition that that record is marked non-processed, and once they + are read (with lock), they are marked processing. When that record + is completed, the indicator is updated to either complete or error. + Many instances of a batch application can be started without any + changes, as the additional column ensures that a record is only + processed once. + + With this option, I/O on the table increased dynamically. In + the case of a updating batch application, this impact is reduced, + as a write will have to occur anyway. + + 7. Extract Table to a Flat File + + This involves the extraction of the table into a file. This + file can then be split into multiple segments and used as input to + the batch instances. + + With this option, the additional overhead of extracting the + table into a file, and splitting it, may cancel out the effect of + multi-streaming. Dynamic configuration can be achieved via changing + the file splitting script. + + 8. Use of a Hashing Column + + This scheme involves the addition of a hash column + (key/index) to the database tables used to retrieve the driver + record. This hash column will have an indicator to determine which + instance of the batch application will process this particular row. + For example, if there are three batch instances to be started, then + an indicator of 'A' will mark that row for processing by + instance 1, an indicator of 'B' will mark that row for + processing by instance 2, etc. + + The procedure used to retrieve the records would then have an + additional WHERE clause to select all rows marked by a particular + indicator. The inserts in this table would involve the addition of + the marker field, which would be defaulted to one of the instances + (e.g. 'A'). + + A simple batch application would be used to update the + indicators such as to redistribute the load between the different + instances. When a sufficiently large number of new rows have been + added, this batch can be run (anytime, except in the batch window) + to redistribute the new rows to other instances. + + Additional instances of the batch application only require + the running of the batch application as above to redistribute the + indicators to cater for a new number of instances. + + 4.2 Database and Application design Principles + + An architecture that supports multi-streamed applications + which run against partitioned database tables using the key column + approach, should include a central streaming repository for storing + streaming parameters. This provides flexibility and ensures + maintainability. The repository will generally consist of a single + table known as the streaming table. + + Information stored in the streaming table will be static and + in general should be maintained by the DBA. The table should + consist of one row of information for each stream of a + multi-streamed application. The table should have a similar layout + to the following table: + + center || Streaming Table || | Program + ID Code Stream Number (Logical ID of the stream) Low Value of the + db key column for this stream High Value of the db key column for + this stream | center + + On program start-up the program id and stream number should + be passed to the application from the architecture (Control + Processing Tasklet). These variables are used to read the streaming + table, to determine what range of data the application is to + process (if a key column approach is used). In addition the stream + number must be used throughout the processing to: + + + + + + Add to the output files/database updates in order for the + merge process to work properly + + + + + + Report normal processing to the batch log and any errors + that occur during execution to the architecture error + handler + + + + + + 4.3 Minimizing Deadlocks When applications run in parallel or + streamed, contention in database resources and deadlocks may occur. + It is critical that the database design team eliminates potential + contention situations as far as possible as part of the database + design. + + Also ensure that the database index tables are designed with + deadlock prevention and performance in mind. + + Deadlocks or hot spots often occur in administration or + architecture tables such as log tables, control tables, lock tables + etc.. The implications of these should be taken into account as + well. A realistic stress test is crucial for identifying the + possible bottlenecks in the architecture. + + To minimize the impact of conflicts on data, the architecture + should provide services such as wait-and-retry intervals when + attaching to a database or when encountering a deadlock. This means + a built-in mechanism to react to certain database return codes and + instead of issuing an immediate error handling, waiting a + predetermined amount of time and retrying the database + operation. + + 4.4 Parameter Passing and Validation + + The streaming architecture should be relatively transparent + to application developers. The architecture should perform all + tasks associated with running the application in a streamed mode + i.e. + + + + + + Retrieve streaming parameters before application + start-up + + + + + + Validate streaming parameters before application + start-up + + + + + + Pass parameters to application at start-up + + + + + + The validation should include checks to ensure that: + + + + + + the application has sufficient streams to cover the whole + data range + + + + + + there are no gaps between streams + + + + + + If the database is partitioned, some additional validation + may be necessary to ensure that a single stream does not span + database partitions. + + Also the architecture should take into consideration the + consolidation of streams. Key questions include: + + + + + + Must all the streams be finished before going into the + next job step? + + + + + + What happens if one of the streams aborts? + + + + + +
+
+ Batch Job Type Specific Concerns + Batch Jobs Types (e.g. conversion, pdf generation, report generation, high volume print, etc.) require different + technologies in the solution space. For example, conversion may require additional XML technologies, Adobe or iText for + PDF generation, different reporting options for report generation, etc. It is often helpful to isolate batch jobs at a + minimum by job type so that required dependencies for batch job types don't pollute other application styles. It is also + key to leaving the application decoupled by style and by execution environment allowing maximum flexibility at deployment + time. + +
+ +
+ diff --git a/docs/src/site/docbook/reference/arch-overview.xml b/docs/src/site/docbook/reference/arch-overview.xml new file mode 100644 index 000000000..b928a11c6 --- /dev/null +++ b/docs/src/site/docbook/reference/arch-overview.xml @@ -0,0 +1,546 @@ + + + + Container Architecture Overview +
+ Introduction + +
+ +
+ Simple Container Architecture Overview + +
+ + Introduction + + This chapter covers the overall spring batch architecture. + The Spring Container Archtiecture is made up of five logical + layers; 1) the Batch Application, 2) the Batch Application Layer, + 3) the batch core layer, and 4) the batch infrastucture + layer. + + + + + + + + + + + Provided + ByLayerDescription + + Application + DeveloperBatch + ApplicationThis is where the + application developer writes their batch jobs and + tasklets. + + Spring Batch Execution + ContainerContainer Application + LayerAllows for extending and + overwriting of the batch support layer for custom + requirements. Facilities implemented in this layer could + migrate down to Batch Support Layer. This is also the layer + to add the project specific jars required by job types + (e.g. reporting jars like Crystal, Brio, etc, form + generation jars like Central Pro or Adobe, + etc). + + Spring Batch Execution + ContainerContainer Support + LayerProvides default + implementations of batch core services including I/O, + Restart, Partitioning, Statistics, and + configurations + + Spring Batch Execution + ContainerContainer Core + LayerEnables configuration, + Common Services & Interfaces, + management + + Spring Batch + InfrastructureBatch-InfrastructureProvides + IO support, Batch style transactions, advanced exception + handling, batch-template, batch-retry + + + + + + + + + + + + Figure 2.0 + + + + - Batch Architecture Layers + + The batch architecture is modeled after a container + architecture, meaning that there are managed resources + essential to high performance batch architectures that are + configured through a spring context. The following sect1s + will provide a quick review of each layer and their role in + the batch architecture. + + + + + + + +
+ +
+ + Batch Applications + + +
+ +
+ + Container Application Layer + + +
+ +
+ + Container Support Layer + + The batch support layer provides default implementations + for all interfaces, interceptors, advice and other core batch + services. Figure 2.3.1 illustrates the following logical + packages. !Batch Support.png! Although physically they break out + into many more than depicted, logically you can think of the + groupings in the following manner: * I/O Support packages * + Restart Support * Lifecycle Support packages * DAO support + layer + + + + + + I/O Support Packages + + The I/O related packages are currently the richest + packages in the batch architecture. They are modeled after + Spring Patterns of Operations and Templates. For example, + you'll see FlatFileInputOperations accompanied with a + FlatFileInputTemplate. The FlatFileInputTemplate is wired up + with a File Descriptor, which contains a Record Descriptor + along with various other properties. With the File and Record + Descriptors the InputTemplate supports a callback method that + allows for the mapping of a record into an object. This + support applies to fixed length records, delimited records + and XML records. To further simplify this a + DefaultFlatFileDataProvider is supplied an input template, + which contains the field and record descriptions, along with + a line mapper that knows how to map the line to an object. + The next() operation on a record simply needs to + readAndMap(lineMapper) a record. This pattern is used over + again for XML and SQL input for simple mapping of input + records to objects. + + In addition to declarative descriptions of the records + that can be re-used by multiple batch jobs, the I/O + facilities also support configurable validation strategies. + The two currently supported are Apache Commons Validator and + Spring's VALang. + + + + + + Restart Support + + The Restart Support provides implementations for a few + common restart strategies that will be discussed further in + the respective sect1. The following are provided + out-of-the-box: * IDList Restart Strategy - a strategy that + supports a batch application where the application does not + have a "process" flag and needs the batch + architecture to track which records have been processed. This + is not the ideal scenario. * Last Processed Restart Strategy + - when the record can be identified through a where and order + by only the last record(s) processed needs to be saved for + restart. * No Restart Strategy - some batch jobs simply + can't support restart. When they are re-run they are + considered to be a new instance of a batch job. * Sql Restart + Strategy - [need some additional javadoc for this + strategy]. + + + + + + Lifecycle Support + + + + + +
+
+
+ + The Core Layer + + The Batch Core interfaces and services are illustrated in + a simplified view of a package diagram. There are roughly seven + logical packages: * Core Spring Extensions * Core Batch Advice * + Core Batch Configuration * Core Batch Repository * Core Batch + Tasklet \ !Batch Core.png! [Figure 2.5] Batch Core Layer + + In the actual physical packaging there are a few more + logical services that the batch execution environment provides. The following + sections will provide an introduction into each set of core batch + facilities. + + + + + + Core Spring Extensions + + The Core Spring extensions provide the scaffolding for + a batch execution environment. This includes facilities for managing the + batch architecture in terms of launching, suspending and + stopping batch jobs. There is house keeping that goes on, + especially in concurrent batch jobs, related to ensuring that + batch jobs quiese properly. The lifecycle management provides + services for the proper initialization and subsequent + shutdown of batch resources and services. The batch + architecture is flexible in terms of how batch jobs may be + launched. For example, batch jobs can be started via JMX + facilities, scripts from the command line that launch a Java + VM. It can also support launching batch jobs through web + services or http. There are no restrictions. Finally, there + are standard batch error codes. These error codes can be + exposed to external utilities, like Schedulers, to ensure + that batch jobs expose the status of jobs to an operational + environment. This is especially important in the batch + context where the modus operandi is headless, meaning + unattended operation. + + + + + + Core Batch Advice + + Core Batch Advice is an inventory of the type of advice + that batch architectures will inject during the runtime of a + batch application. These are defined as a set of extensible + interfaces, with a number of default implementations in the + support layer that provide some of the most common types of + advice. Partition Advice is helpful with large datasets that + need to be "chunked" up and run concurrently for + better through put. Resource Advice is helpful for + registering interest in transactional information so that + file locations can be kept in sync with information processed + within a transaction. In addition, the resource is associated + with the correct step context and its associated + configuration properties. Skip advice is applied for records + that the tasklet is unable to process. Restart Advice is + helpful for Restartable jobs where for advising the job on + how to restart. There is considerable variability on how + restart can occur. For example, a job may be marking records + as "processed" and the restart advice will advise + the process with query that restarts the job at the last + successfully processed record. Finally, Statistics are vital + in operational environments to report on records processed, + records skipped and total number of records read. In + addition, certain batch jobs lend themselves to custom + reporting to expose additional business level information + like the number of trades processed or cases opened, + etc. + + + + + + Core Batch Configuration + + Batch configuration is considerably different from + online web applications or SOA based applications. The Core + Batch Configuration provides a place for configuring runtime + properties related to the batch application style. This + includes the ability to add Commit Policy. In a batch style + application it is often advantageous to keep the commit + interval as high as possible when processing Logical Units of + Work. Whereas in an online web application with declarative + transaction the transaction scope would be at the entrance to + a business service, a batch transaction scope may include + many logical units of work before a transaction commit is + executed. A Start Policy allows a configuration to tell the + batch job whether it is Restartable, and if so, what type of + restart to initiate. Some jobs are not restartable and care + should be taken to ensure that information is not applied + multiple times when the business rules do not allow for it. + Exception policies deal with what to do when exceptions + occur. This impacts logging policies and exception handling. + The architecture defines a common set of exceptions that + projects can apply handlers to like processing errors, + validation errors, parsing errors, missing configuration + parameters, etc. + + + + + + 2.5.4 Container Repository + + This is an internal package for storing the state of a + batch job and any associated partition and step status. + + + + + + Core Batch Tasklet + + The core batch tasklet is where control is handed off to + the application. There are a number of patterns that have + been observed in processing batch data. Spring Core Batch + Tasklet implements the most common patterns and provides and + extension point for additional Tasklet processing + implementations. The basic idea of tasklet provides the + facilities for reading and processing data. The simplest + implementation of Tasklet, the ReadProcessTasklet, handles both + the input and output of data within one class. An alternative + implementation, the DataProviderProcessTasklet, provides + functionality for 'split processing'. This type of + processing is characterized by separating the reading and + processing of batch data into two seperate classes: + DataProvider and TaskletProcessor. The DataProvider class + provides a solid means for reusablility and enforces good + architecture practices. Because an object *must* be returned + by the DataProider to continue processing, (Returning null + indicates processing should end) a developer is forced to + read in all relevant data, place it into domain or value + objects, and return the object. The TaskletProcessor will then + use this object within the business logic and final + output. + + + + + +
+ +
+ + Container's Use of batch + infrastructure + + + + + + Infrastructure Provided I/O + + The I/O core interfaces and implementations provide + facilities for simplifying the extraction of data from I/O + sources like files and database tables. The key concepts are + FieldDescriptors and FieldSets along with appropriate CallBack + Handlers. These are modeled after common spring operations and + templates like JdbcTemplate. Through the use of LineMappers a + developer needs only to describe a record format and write the + appropriate callback method that maps the parsed record into an + object of their choice. These can either be true POJO objects + or Value Objects (structures) that are subsequently available + for the tasklet to processs. The interface for Field Descriptors + also allows for a level of validation through the use of + Spring's VALang or Apache's Common Validator. + + + + + + Core Batch Interceptors & Interceptor + Services + + + + + + Batch Operations & Batch Template + + Interceptors and the associated services are the key + to how advise is applied in the batch architecture. The + interceptors are Point Cuts in the batch lifecycle that + allow the injection of advise. The shared lifecycle + behavior abstracted through the BatchLifeCycleInterceptor + defineds three methods; init, onError and finalize. All + subclasses of LifeCycleInterceptor define default behavior + for these three methods. The JobLifecycleInterceptor + further exposes the methods beforeJob(), beforeStep(), + afterJob(), and afterStep() allowing hooks into the + lifecycle for specific advise. The Batch Architecture + provides default implementations for all lifecycle point + cuts, or interception points. The Tasklet Interceptor, in + addition to the standard lifecycle methods, implements + logic around beforeLuw(), afterLuw(), + commitIntervalStarted() and commitIntervalCompleted(). + Having well defined lifecycle interception points allows + for the easy insertion of custom advice into the batch + runtime environment. + + + + + + + + + +
+ +
+ + Batch Execution Container Configurations + + In addition to core facilities for configuring or wiring + together jobs and steps with required resources, policies, and + interceptors, spring batch allows considerable flexibility in how + scalability is achieved. More options for scalability will be + available in the future. The important key for scalability in Java + is the recognition that there is a limit to what one JVM may scale + up to in terms of number of threads, managed resources, memory + configuration, etc. The spring batch architecture allows for the + configuration of simple batch jobs where one VM and one process is + sufficient to do perform the work within a batch window all the way + through many threads distributed within a cluster of JEE servers. + The figure below illistrates the scalability spectrum. + + This is not to be understood as the only way to scale batch + jobs as there are many factors. For example, other federated java + architectures hold potential like Teracotta or Gigaspaces although + there is no current implementation for these distributed models in + the current batch architecture. !scalability-model.png! [Figure + 2.3.1] - Scalability Model + +
+ + Single VM Simple Batch Execution + Container - One Job, One Step, One Partition + + The simplest configuration is one job with with step and + hence, one implied partition. Implied means that there is nothing + for the developer to consider because the default number of + partitions is one. There is typically one input source and one + output source in this simple configuration. See the Simple Tasklet + Job for an example of what this configuration looks like. A + simple configuration still typically configures a datasource + context, the batch configuration for describing the Job, Step, + along with the associated configured policies, field descriptors, + and line mappers. !SimpleTradeConfiguration.jpg! [Figure 2.3.1] + Simple Container Configuration + + The details of this configuration will be covered + thoroughly in subsequent sect1s of the document but for now it + should be understood that Job, the Step, the input template, the + file descriptor with its associated line mapper, and the output + (e.g. the TradeWriter). + +
+ +
+ + Single VM Multi-threaded Batch Execution + Container Configuration - One Job, One Step, Multiple + Partitions + + In a Single JVM using partitioning a multi-threaded + execution is supported. [This is still work in progress] + +
+ +
+ + Batch Execution Container Hosted in J2EE + Container - managed environment + + The J2EE container model has fallen under fire over the + past few years for many valid reasons. There are some things that + the J2EE container do very well though that projects should + consider when planning for scalability with batch architectures. + Commercial and open source containers like WebSphere, BEA and + JBOSS typically: + + + + + + manage datasources effectively along with attendent + services like prepared statement caching. + + + + + + manage transactions effectively including many + configurable properties for long lived transactions. + + + + + + manage thread pools more effectively. + + + + + + supply robust implementations of JTA, a requirement + when batch jobs output to multiple XA resources like JMS and + JDBC. + + + + + + manage distribution effectively including domains, + clusters and cells + + + + + + provide robust JMX management for configuring, managing + and administering distributed applications. + + + + + + workload management facilities (clusters) provided by + J2EE vendors + + + + + + Projects are encouraged to deploy batch applications with + the simplest configuration possible, but when federated JVMs are + a requirement to process volumes of data within a batch window, + batch-in-container provides an effective way of distributing the + processing. Spring Batch supports this through a simple change in + configuration. [Work in progress on the exact implementation - + being released as part of M2]. + +
+
+ +
+ diff --git a/docs/src/site/docbook/reference/batch-job-testing.xml b/docs/src/site/docbook/reference/batch-job-testing.xml new file mode 100644 index 000000000..b5d663308 --- /dev/null +++ b/docs/src/site/docbook/reference/batch-job-testing.xml @@ -0,0 +1,17 @@ + + + Batch Unit and Integration Tests + +
+ Unit Testing + Document Batch Job Unit Testing features. This includes the use of Mock Objects, + embedded database (HSQLDB), etc. +
+ +
+ Integration Testing + Document how to test against the targeted database, applications, etc. +
+ +
diff --git a/docs/src/site/docbook/reference/batch-launch.xml b/docs/src/site/docbook/reference/batch-launch.xml new file mode 100644 index 000000000..6895fa5db --- /dev/null +++ b/docs/src/site/docbook/reference/batch-launch.xml @@ -0,0 +1,17 @@ + + + Run Tier - Launching Batch Jobs +
+ Mapping Batch Error Codes to Launch Client Error Codes + Mapping Batch Error Codes to Launch Client Error Codes +
+
+ Launch Batch from Command Line + Document Command Line Launching +
+
+ Launch Batch On Demand + Document Launching Batch Jobs on Demand +
+
diff --git a/docs/src/site/docbook/reference/batch-performance-testing.xml b/docs/src/site/docbook/reference/batch-performance-testing.xml new file mode 100644 index 000000000..8049673de --- /dev/null +++ b/docs/src/site/docbook/reference/batch-performance-testing.xml @@ -0,0 +1,52 @@ + + + Batch Performance Testing +
+ Performance Testing Overview + A batch performance test team needs to have the following at their disposal: + + + + + Define performance Targets - + + + Establishing the requirements for a performance testing environment - + + + Performance Data - generating adequate volumes of realistic data for performance testing + + + Performance Tools - + + + Performance Team Roles - Tool SME's, performance DBA. + + + +
+ +
+ Defining Performance Targets + +
+ +
+ Establishing Performance requirements and installing the environment. + Establishing the requirements for the performance environment. +
+ +
+ Performance Data + Generating adequate volumes of realistic data for performance testing. +
+
+ Performance Tools + +
+
+ Performance Team Roles + Tool SME's, performance dba's, environment experts (OS, JVM, etc.) +
+
diff --git a/docs/src/site/docbook/reference/container-overview.xml b/docs/src/site/docbook/reference/container-overview.xml new file mode 100644 index 000000000..cb8d71297 --- /dev/null +++ b/docs/src/site/docbook/reference/container-overview.xml @@ -0,0 +1,272 @@ + + + Overview of the Spring Batch Environment + +
+ + + Overview of the Spring Batch Simple Batch Execution + Environment + + + + The diagram below provides an overview of the high level + components, technical services, and basic operations + required by a batch architecture. This architecture + framework is a blueprint that has been proven through + decades of implementations on the last several generations + of platforms (COBOL/Mainframe, C++/Unix, and now + Java/anywhere). The Simple Batch Execution Environment + provides a physical implementation of the layers, components + and technical services commonly found in robust, + maintainable systems used to address the creation of simple + to complex batch applications, with the infrastructure and + extensions to address very complex processing needs. The + materials below will walk through the details of the + diagram. + + +
+ +
+ + + Simple Batch Execution Environment high level flow and + interaction of the architecture. + + + + + + + + + + + Figure 1: Batch Execution Environment + + + + + Tiers The application style is organized into four logical + tiers, which include Run, Job, Application, and Data tiers. + The primary goal for organizing an application according to + the tiers is to embed what is known as "separation of + concerns" within the system. Effective separation of + concerns results in reducing the impact of change to the + system. + + + + + + + + Run Tier: + The Run Tier is concerned with the scheduling and + launching of the application. A vendor product is + typically used in this tier to allow time-based and + interdependent scheduling of batch jobs as well as + providing parallel processing capabilities. + + + + + + + + Job Tier: + The Job Tier is responsible for the overall + execution of a batch job. It sequentially executes + batch steps, ensuring that all steps are in the + correct state and all appropriate policies are + enforced. + + + + + + + + Application Tier: + The Application Tier contains components required to + execute the program. It contains specific tasklets + that address the required batch functionality and + enforces policies around a tasklet execution (e.g., + commit intervals, capture of statistics, etc.) + + + + + + + + Data Tier: + The Data Tier provides the integration with the + physical data sources that might include databases, + files, or queues. + Note + : In some cases the Job tier can be completely + missing and in other cases one Job Script can start + several Batch Job instances. + + + + + + +
+ +
+ + High Level Processing Flow + + + The diagram above illustrates the flow and architecture + components in a typical batch run execution. + + + Standard interaction is described as follows: + + + 1. + In the Run tier, a Scheduler starts a batch application by + invoking a Job Script. The Scheduler identifies what batch + process it wants to run by passing the name of the batch + process and any required additional parameters to the Job + Script. + + + + 2. + The Job Script initializes the program and executes any job + specific scripts prior to calling the Batch Launcher. + + + + 3. + The Batch Launcher starts the Batch Execution Environment + based upon any environment settings established in the + script. + + + + 3.1 + The Batch Environment starts and controls the batch execution. + It initializes the Job execution environment with static + configuration items such as database settings, logging + levels and creates a Job based on the Job Configuration + created by a Batch Developer. + + + + 4 + Based on configuration provided by a Batch Developer, the + Job sequentially executes steps after checking policies to + ensure that each step should be started. The status of the + job and step (start time, end time, status such as + "started" or "completed") is stored at + various points during the process. + + + + 5.1 + In order to maintain data integrity, at the application + tier, the Step acts as a controller to ensure that either an + entire group of actions completes successfully or that none + of the actions completes. This group of actions is referred + to as a logical unit of work (LUW). The Step controls the + overall execution of the Tasklet, ensuring that transaction + are committed at the appropriate time, and restart and + statistics information is stored appropriately. The first + thing the Step is responsible for is the initialization of + the data required to begin processing. The Step will + interact with other architecture components, such as the + Input Source, to setup the data required to be processed. + + + + 5.1.1 + The Input Source provides services to access various data + sources. It provides location transparency to the Batch + Tasklet and hides the physical location details of the data. + + + + 5.2 + Once the data is initialized by the Input Source, the Step + will call into the Tasklet to begin processing. The Tasklet + contains the business logic to define the LUW and the Step + repeatedly calls the Tasklets LUW to finish the business + function. The Step does this by first invoking the execute + method on the Tasklet in order to acquire a single + record/set of data for processing. + + + + 5.2.1 + Before a record is returned to the Tasklet, it may be + validated by any number of validation Frameworks that can be + provided to an input source. A single record/set of data is + gathered by interacting with the Input Source. + + + + 5.3 + Once a record/set has been obtained, the step calls the + tasklet to begin processing. + + + + 5.3.1 + The Tasklet executes its internal business logic by calling + other Business Logic components as necessary. Based on the + business service, it requests or persists objects from the + data access components. + + + + 5.3.3 + Data Access components can be leveraged retrieve or persist + domain objects. + + + + 5.3.4 + Once the business logic has been executed, the resulting + output record is written out by utilizing the Output Source + interface. The Step will repeatedly call steps 5.2 -> 5.3 + for every record provided by the Input Source. + + + + 5.4 + Once all of the records are processed, the Step calls the + Tasklet to perform any clean up activities such as closing + connections, exporting files, etc. + + + + 5.4.1 + The Step is responsible for committing data associated with + the remaining logical units of work as well as performing + any finalization and administrative functions (e.g. closing + database connections). + + + + Once the Step has completed finalization the control is + passed back to the Job, where any necessary logging or clean + up is executed for application termination and wrap-up -- + provided there are no additional Steps to execute. + + +
+ +
+ diff --git a/docs/src/site/docbook/reference/core.xml b/docs/src/site/docbook/reference/core.xml new file mode 100644 index 000000000..9b65e9375 --- /dev/null +++ b/docs/src/site/docbook/reference/core.xml @@ -0,0 +1,448 @@ + + + + Spring Batch Core - the Domain language of Batch + +
+ Introduction + + To any experienced batch architect, the overall concepts of batch + processing described above should be familiar and comfortable. There are + “Jobs” and “Steps” and a developer supplied processing unit that Spring + Batch refers to as the “Tasklet.” The following diagram is only a slight + variation of the batch reference architecture that has been used for + decades. JCL and COBOL developers are likely to be as comfortable with the + concepts as C++, C# and Java developers. However, because of the Spring + patterns, operations, templates, callbacks, and idioms, there are + opportunities for + + significant improvement in adherence to a clear separation of + concerns, + + + + clearly delineated architectural layers and services provided + as interfaces, + + + + simple and default implementations that allowed for quick + adoption and ease of use out-of-the-box, and + + + + significantly enhanced extensibility. + + + + The diagram below provides an overview of the high level components, + technical services, and basic operations required by a batch architecture. + This architecture framework is a blueprint that has been proven through + decades of implementations on the last several generations of platforms + (COBOL/Mainframe, C++/Unix, and now Java/anywhere). The Simple Batch + Execution Environment provides a physical implementation of the layers, + components and technical services commonly found in robust, maintainable + systems used to address the creation of simple to complex batch + applications, with the infrastructure and extensions to address very + complex processing needs. The materials below will walk through the + details of the diagram. +
+ +
+ Simple Batch Execution Environment high level flow and + interaction of the architecture. + + + + + + + + + + + Figure 1: Batch Stereotypes + + + The application style is organized into four logical tiers, which + include Run, Job, Application, and Data tiers. The primary goal for + organizing an application according to the tiers is to embed what is known + as "separation of concerns" within the system. These tiers can be + conceptual but may they prove effective in mapping the deployment of the + artifacts onto physical components like Java runtimes and integration with + data sources and targets. Effective separation of concerns results in + reducing the impact of change to the system. The four conceptual tiers + containing batch artifacts are: + + + + Run Tier: The Run Tier is + concerned with the scheduling and launching of the application. A + vendor product is typically used in this tier to allow time-based + and interdependent scheduling of batch jobs as well as providing + parallel processing capabilities. + + + + Job Tier: The Job Tier is + responsible for the overall execution of a batch job. It + sequentially executes batch steps, ensuring that all steps are in + the correct state and all appropriate policies are enforced. + + + + Application Tier: The + Application Tier contains components required to execute the + program. It contains specific tasklets that address the required + batch functionality and enforces policies around a tasklet execution + (e.g., commit intervals, capture of statistics, etc.) + + + + Data Tier: The Data Tier + provides the integration with the physical data sources that might + include databases, files, or queues. Note : In some cases the Job tier can be + completely missing and in other cases one Job Script can start + several Batch Job instances. + + In addition the components describe the batch interaction + and services stereotypes that are the domain language and interfaces + implemented by developers in constructing a batch solution. As he diagram + illustrates, custom applicaton archifacts, generally created by the + developer, are the following: + + + + Job Scripts + + + + JobConfigurations + + + + Tasklet + + + + Business Logic + + + + The application architect needs to consider the batch execution + environment with the following issues: + + + + Define how the batch jobs will be launched + + + + Define the Job Execution Environment + + + + Step construction and Configuration + + + + ItemReaders + + + + ItemWriters + + + + Data Access Strategies + + + + The grey icons indicate the technologies selected as part of the + batch solution that are not part of the final solution and entail items + like: + + + + Schedulers (e.g. Quartz, Tivoli, etc.) + + + + Physical Resources in the Data Tier that are the source and + target of ItemReaders and Writers like Message Queues, Databases, + Files and Print Queues. + + + +
+ Batch Domain Stereotypes + We will discuss each of these Batch domain Stereotypes + individually. This section describes stereotypes relating to the concept of a batch job. + A job is an entity that encapsulates an entire batch process. +
+ Job Configuration + + The job configuration could be described as the heart of the Spring Batch framework. + It is represented by a Spring bean of class _JobConfiguration_ and contains all of + the information necessary to define the operations performed by a job. A job configuration + is typically contained within a Spring XML configuration file and the job's name is + determined by the "id" attribute associated with the job configuration bean. The job configuration contains: + + + + The simple name of the job + + + Definition and ordering of [Step Configurations|#Step Configuration] + + + The limit of how many times this job may be started + + + Whether or not the job is restartable + + + The mechanics of defining a job configuration will be discussed in the next chapter. [Provide a link] +
+
+ Job Instance + A job instance refers to the business concept of a single job invocation. In other words, suppose + you have a job called "foo" that is run three times a day. There will be one "foo" configuration, + and each time "foo" is supposed to run would be an instance of the "foo" job. Each instance would be + uniquely identified as each one represents a distinct batch need. Further, each instance might + have attempted several times to complete its work. Each attempt is represented by a [#Job Execution], + described below. A job instance is not considered to be complete until an associated job execution + completes successfully. As such, a single job instance may have many executions. + + For example, a unique instance might be identified by just a job name, or by the combination of a + job name and a scheduled date. Using this second type of identification, we might have two distinct + instances, "foo-01-01-2008" and "foo-01-02-2008." Although these two instances would share the same + configuration, they would each have their own set of executions and the successful completion of one + instance would not affect the status of the other. + + Job instances are represented by objects of the _JobInstance_ class, which are created when the + job is executed. Each job instance contains references to related [Step Instances|#Step Instance] + and a _JobIdentifier_ that uniquely identifies this job instance. + + +
+
+ Job Execution + +A job execution refers to the technical concept of a single attempt to run a job. It is a single attempt to execute the logic represented by a job instance. A job execution may end in failure or success, but the job instance corresponding to a given execution will not be marked as complete unless the execution completes successfully. + + +For instance, if we have a job instance "foo-01-01-2008" that fails to successfully complete its work the first time it is run, when we attempt to run it again, a new job execution will be created. If our "foo" configuration is restartable, we may begin our second job execution from a restart point. Otherwise, our job execution will start from the beginning. In either case, we will see that our single job instance has had two job executions. + + +Job executions are represented by objects of the _JobExecution_ class. These job executions are created by an implementation of the _JobExecutorFacade_ interface from a given _JobInstance_ corresponding to a unique _JobIdentifier_. Each job execution contains a reference to its corresponding job instance, related Step Executions and step/chunk context data. + +
+
+ Step Stereotypes + +This section describes stereotypes relating to the concept of a batch step. A step is an entity that encapsulates a single, independent phase of a batch job. Therefore, every batch job is composed entirely of one or more batch steps. + +
+
+ Step Configuration + +The step configuration contains all of the information necessary to define a discrete set of business logic within a job configuration. This is a necessarily vague description because the contents of any given step configuration are at the discretion of the developer writing your jobs. A step can be as narrowly defined as a single line of code or as broadly defined as necessary to complete the entire work of your job. There are several factors that will affect the breadth of your step configurations. + + + + Re-usability - step definitions can be shared between jobs + + + Transaction Management - depending on your desired transaction strategy, you may divide the work of your job differently between steps + + + Extensibility - adequately granular definition of steps allows the addition or subtraction of steps at a later time in the appropriate position within your job configuration + + + +Step configurations are defined by instantiating implementations of the _StepConfiguration_ interface. Additionally, the utility class _StepConfigurationSupport_ provides a basic implementation of _StepConfiguration_ with default functionality that should be common to any concrete _StepConfiguration_ implementation. Generally, all step configuration implementations should extend from this class. + + +Two step configuration classes are available in the Spring Batch framework, and they are each discussed in detail in other sections of this guide. For most situations, the _SimpleStepConfiguration_ implementation is sufficient, but custom transaction management behavior can also be configured by using a _RepeatOperationsStepConfiguration_. + +
+
+ Step Instance + +A step instance, represented by the _StepInstance_ class, represents the business concept of a single step within a job invocation. That is to say, every job instance contains one or more step instances. + + +For example, suppose we have a job instance called "foo-01-01-2008" that is an instance of a job configuration containing three steps. Suppose these steps are named "step1", "step2" and "step3." There will be corresponding "foo-01-01-2008#step1", "foo-01-01-2008#step2" and "foo-01-01-2008#step3" step instances, which will be distinct from the step instances of any other job instance (e.g. those of "foo-01-02-2008"). +{note}The step instance naming here is for clarity, this is not necessarily how the instance will be named internally within the framework.{note} + + +Each step instance will contain the current status of the batch execution, restart data, and a reference to its corresponding [Job Instance]. Additionally, a step instance keeps track of how many attempts are made to run the corresponding step. Each attempt to run a step will create a [Step Execution], so a single job instance might have several corresponding step executions. + +
+
+ Step Execution + +A step execution represents the technical concept of a single attempt to execute a step. It is a single attempt to execute the logic represented by a step instance. + + +For instance, if we have a step instance "foo-01-01-2008#step1" that fails to successfully complete its work the first time it is run, when we attempt to run it again, a new step execution will be created. Each of these step executions may represent a different invocation of the batch framework, but they will all correspond to the same step instance. + + +Step executions are represented by objects of the _StepExecution_ class. These step executions are created by an implementation of the _JobExecutor_ interface from a given _StepInstance_ and _JobExecution_. Each step execution contains a reference to its corresponding step instance and job execution, and transaction related data such as commit and rollback counts, start and end times and a _Properties_ instance containing statistics. + +
+
+ Tasklets + + A tasklet represents the execution of a logical unit of work, as defined by its implementation of the Spring Batch provided _Tasklet_ interface. Tasklets are used when defining step configurations to specify the work done by the step. Subsequently, the logic in a tasklet is atomic in terms of transactions. A transaction will never commit until an entire tasklet execution is complete (unless an exception occurs - a transaction might either commit or rollback if that behavior is specified in the step's exception management strategy). + +
+
+ Item-Oriented Processing Stereotypes + +A powerful batch processing paradigm implemented by the Spring Batch framework is the concept of item-oriented processing. That is, doing work by defining each unit of work as the operation of retrieving an item from input and then processing that item, including any side effects that processing might entail, such as file or database operations. + + +There are two basic stereotypes that represent the first-class participants in item-oriented processing, item providers and item processors. They are each represented by a simple interface provided by the Spring Batch framework, which allows free reign over their implementations and improves our ability to leverage the Spring framework's dependency injection capabilities. + +
+ Item Readers + + An item reader is an object that is used to retrieve the inputs for a step, one at a time. When the item reader has exhausted the items it can provide, it will indicate this in a meaningful way (generally by returning _null_). When coupled with an item processor, this forms a complete item-oriented process, as each item taken from the provider is then processed by the processor. + +
+
+ Item Writers/Processors + + An item processor is an object that is used to perform processing for a step, one item at a time. Generally, an item processor has no knowledge of the input it will receive next, only the item that that was passed in its current invocation. As a result, item processors will generally make no assumptions about the input they receive an treat every item the same way and keep track of its own state between invocations. When coupled with an item provider, this forms a complete item-oriented process, as each item taken from the provider is then processed by the processor. + +
+
+
+ Support Stereotypes + While item providers and processors serve as the main entry points for item-oriented processing, they are supplemented by a number of support classes that perform specific tasks within the provider / processor lifecycle. These support stereotypes are useful for dividing the work of item providers and processors into reusable pieces, as well as abstracting away the details of processing, such as interaction with external systems. Additionally, they give us another opportunity to leverage the powerful configuration features of the Spring framework, as we can switch between several beans implementing these support interfaces without changing the driving item provider or processor. + +
+ Input Sources + An input source is a class that mediates interactions with an external source of input data, such as a file or a database. An input source often serves as a support for an item provider, typically abstracting away the details of interaction, such as the creation and maintenance of file handles, sockets or database connections. + +
+
+ Item Transformers + An item transformer is a class that is capable of taking an object and changing it somehow before processing occurs. For instance, an item transformer my alter an object by changing its properties or by replacing it with another object entirely, such as a wrapper or derivative object. It can also be defined as an adaptor, allowing an object of one type to be converted for use as an object of a second type. + +
+
+ Item Writers + An item writer is a class that mediates interactions with an external target of output data, such as a file or database. An item writer often serves as a support for an item processor, typically abstracting away the details of interaction, such as the creation and maintenance of file handles, sockets, database connections and other output-related tasks such as buffering and stream flushing. + +
+
+ +
+ High Level Processing Flow + + The diagram above illustrates the flow and architecture components + in a typical batch run execution. + + Standard interaction is described as follows: + + 1. In the Run tier, a Scheduler + starts a batch application by invoking a Job Script. The Scheduler + identifies what batch process it wants to run by passing the name of the + batch process and any required additional parameters to the Job + Script. + + 2. The Job Script initializes the + program and executes any job specific scripts prior to calling the Batch + Launcher. + + 3. The Batch Launcher starts the + Batch Execution Environment based upon any environment settings + established in the script. + + 3.1 The Batch Environment starts + and controls the batch execution. It initializes the Job execution + environment with static configuration items such as database settings, + logging levels and creates a Job based on the Job Configuration created by + a Batch Developer. + + 4 Based on configuration provided + by a Batch Developer, the Job sequentially executes steps after checking + policies to ensure that each step should be started. The status of the Job + and Step (start time, end time, status such as "started" or "completed" is + stored at various points during the process. In order to maintain data + integrity at the application tier, the Step acts as a controller to ensure + that either an entire group of actions completes successfully or that none + of the actions completes. In online applications the Unit Of Work and the + scope of a transaction tend to be the same thing (e.g. update customer). + This group of actions controlled by a user interaction is referred to as a + logical unit of work (LUW). However, in batch processing the Step + frequently separates the transactional scope from the LUW so that many + LUWs complete within one commit. This greatly improves batch throughput + (see pseudo code above where REPEAT(size=500). The Step controls the + overall execution of the Tasklet, ensuring that transactions are committed + at the appropriate time and that restart and statistics information is + stored appropriately. + + 4.1 The first thing the Step is + responsible for is the initialization of the data required to begin + processing. The Step will interact with other architecture components, + such as the Input Source, to setup the data required to be + processed. + + 4.1.1 The Input Source provides + services to access various data sources. It provides location transparency + to the Batch Tasklet and hides the physical location details of the + data. + + 4.2 Once the data is initialized by + the Input Source, the Step will call into the Tasklet to begin processing. + The Tasklet contains the business logic to define the LUW and the Step + repeatedly calls the Tasklets LUW to finish the business function. The + Step does this by first invoking the execute method on the Tasklet in + order to acquire a single record/set of data for processing. + + 4.2.1 Before a record is returned + to the Tasklet, it may be validated by any number of validation Frameworks + that can be provided to an input source. A single record/set of data is + gathered by interacting with the Input Source. + + 4.3 Once a record/set has been + obtained, the step calls the tasklet to begin processing. + + 4.3.1 The Tasklet executes its + internal business logic by calling other Business Logic components as + necessary. Based on the business service, it requests or persists objects + from the data access components. + + 4.3.3 Data Access components can be + leveraged retrieve or persist domain objects. + + 4.3.4 Once the business logic has + been executed, the resulting output record is written out by utilizing the + Output Source interface. The Step will repeatedly call steps 4.2 -> 4.3 + for every record provided by the Input Source. + + 4.4 Once all of the records are + processed, the Step calls the Tasklet to perform any clean up activities + such as closing connections, exporting files, etc. + + 4.4.1 The Step is responsible for + committing data associated with the remaining logical units of work as + well as performing any finalization and administrative functions (e.g. + closing database connections). + + Once the Step has completed finalization the control is passed back + to the Job, where any necessary logging or clean up is executed for + application termination and wrap-up -- provided there are no additional + Steps to execute. +
+
diff --git a/docs/src/site/docbook/reference/execution.xml b/docs/src/site/docbook/reference/execution.xml new file mode 100644 index 000000000..c77ca4fd9 --- /dev/null +++ b/docs/src/site/docbook/reference/execution.xml @@ -0,0 +1,99 @@ + + + + The Batch Execution Environment + +
+ Introduction + + The "execution" layer is fertile ground for collaboration and + contributions from the community and from projects in the field. There is + lifecycle support for starting and stopping jobs. The vision for this is + that there can be multiple implementations of this interface providing + different architectural patterns, and delivering different levels of + scalability and robustness, without changing either the business logic or + the job configuration. The initial 1.0 release of Spring Batch will have a + single implementation for the Simple Batch Execution Service. + + The Execution Environment is responsible for providing + implementations of the core domain concepts. This includes: + + Run Tier - Implement the bootstraping and launching of the + Execution Environment. + + + + Job Tier - Implement the Job Configuration and Job Execution + strategies. + + + + Application Tier - Impleement the Step Configuration and Step + Executor strategies. + + + + Data Tier - Implement a Repository solution for storing the + persistence state of the batch domain. + + We will describe the flow of the simple batch + execution envrionment that is provided with spring-batch to clarify the + sequence of processing in the batch environment. The simple batch + execution environment is a concrete implementation of the core interfaces. + The Simple Batch Execution Environment can be described in the following + way. + + In the Run Tier, asingle Java Project may have one or more + Jobs. Jobs are configured with the JobConfiguration bean. + + + + A single Job must contain at least one Step.Steps are + configured with the StepConfiguration bean. If more than one step is + configured for a Job, then they are executed serially and after the + previous step (all of its items / records) is complete. + + + + Steps are executed by StepExecutor classes, being the + DefaultStepExecutor the most simple for it. + + + + Steps in turn are divided into two “cycles” or “iterators”; + the “stepOperations” iterator and the “chunkOperations” iterator. + The StepOperations of the Step cycles while there is data to be + processed (i.e. a CompletionPolicy returns true) whereas the + chunkOperations executes for every cycle of the stepOperations and + it is used as Transaction controlling mechanism. The idea is that + the chunkOperations iterator executes the number of “commit size” + defined in the configuration unless another CompletionPolicy is + defined. After “# of records per Commit” cycles, it returns control + to the StepOperations which in turn calls again the ChunkOperations + if there are more records to be processed. Programmers should only + define chunkOperations and not stepOperations, unless a parallel + execution strategy is chosen. + + + + The RepeatTemplate class (implementator of the chunk and step + Operations) requires a Tasklet. A Tasklet is responsible for doing + something in each cycle of the "stepOperations". Simple tasklets may + fetch and process data in the same class but typicallyIdevelopers + will use the ItemProviderProcessTasklet that separates fetching data + from processing it in the ItemProvider and ItemProcessor interfaces. + If using DB access it's also common to use an + InputSourceItemProvider as ItemProvider and an + OutputSourceItemProcessor as ItemProcessor. There are even more + specialized classes for working the Jdbc (databases). + + +
+ +
+ Simple Batch Execution Environment + + +
+
\ No newline at end of file diff --git a/docs/src/site/docbook/reference/glossary.xml b/docs/src/site/docbook/reference/glossary.xml new file mode 100644 index 000000000..fdd13231f --- /dev/null +++ b/docs/src/site/docbook/reference/glossary.xml @@ -0,0 +1,184 @@ + + + + Glossary +
+ Glossary Items + + + + Batch: An accumulation of + business transactions over time. + + + Batch Application Style: + Term used to designate batch as an application style in its own + right similar to online, Web or SOA. It has standard elements of + input, validation, transformation of information to business + model, business processing and output. In addition, it requires + monitoring at a macro level. + + + Batch Processing: The + handling of a batch of many business transactions that have + accumulated over a period of time (e.g. an hour, day, week, + month, or year). It is the application of a process, or set of + processes, to many data entities or objects in a repetitive and + predictable fashion with either no manual element, or a separate + manual element for error processing. + + + + Batch Window: The time + frame within which a batch job must complete. This can be + constrained by other systems coming online, other dependent jobs + needing to execute or other factors specific to the batch + environment. + + + + + + Step Controller: It is the + main batch task or Unit of Work controller. It initializes the + tasklet, and controls the transaction environment based on commit + interval setting, etc. + + + + + + Tasklet: The main + application program created by application developer to process + the business logic for each LUW. + + + + + + Batch Job Type: Job Types + describe application of jobs for particular type of processing. + Common areas are interface processing (typically flat files), + forms processing (either for online pdf generation or print + formats), report processing. s + + + + + + Driving Query: A driving + query identifies the set of work for a job to do; the job then + breaks that work into individual units of work. For instance, + identify all financial transactions that have a status of + "pending transmission" and send them to our partner + system. The driving query returns a set of record IDs to process; + each record ID then becomes a unit of work. A driving query may + involve a join (if the criteria for selection falls across two or + more tables) or it may work with a single table. + + + + + + Logicial Unit of Work + (LUW): A batch job iterates through a driving query + (or another input source such as a file) to perform the set of + work that the job must accomplish. Each iteration of work + performed is a unit of work. + + + + + + Commit Interval: A set of + LUWs constitute a commit interval. + + + + + + Partitioning: Splitting a + job into multiple threads where each thread is responsible for a + subset of the overall data to be processed. The threads of + execution may be within the same JVM or they may span JVMs in a + clustered environment that supports workload balancing. + + + + + + Staging Table: A table + that holds temporary data while it is being processed. + + + + + + Restartable: - a job that + can be executed again and will assume the same identity as when + run initially. In othewords, it is has the same job instance + id. + + + + + + Rerunnable - a job that is restartable and manages it's own state in terms of previous run's record + processing. Note>>: Rerunnable is tied to the driving query. If the driving query can be formed so that it will limit the + processed rows when the job is restarted than re-runnable = true. This is managed by the application architecture. Often times a + condition is added to the where statement to limit the rows returned by the driving query with something like "and + processedFlag != true". + + + + + + + Repeat: One of the most basic units of batch processing, that defines repeatability calling a + portion of code until it is finished, and while there is no error. Typically a batch process would be repeatable as long as there is input. + + + + + + + Retry: Simplifies the execution of operations with retry semantics most frequently associated + with handling transactional output exceptions. Retry is slightly different from repeat, rather than continually calling a block of code, + retry is stateful, and continually calls the same block of code with the same input, until it either succeeds, or some type of retry limit + has been exceeded. It is only generally useful if the operation is non-deterministic meaning that a retry on a subsequent invocation might + succeed because something in the environment has improved. + + + + + + + Recover: Recover operations handle an exception in such a way that a repeat process is able to + continue. + + + + + + Skip: Skip is a recovery strategy often used on file input sources as the strategy for ignoring + bad input records that failed validation. + + + + +
+
+ diff --git a/docs/src/site/docbook/reference/index.xml b/docs/src/site/docbook/reference/index.xml new file mode 100644 index 000000000..a3334e686 --- /dev/null +++ b/docs/src/site/docbook/reference/index.xml @@ -0,0 +1,42 @@ + + + + + Spring Batch - Reference Documentation + Spring Batch 1.0 + + + Dave + Syer + + + Wayne + Lund + + + Scott + Wintermute + + + + + Copies of this document may be made for your own use and + for distribution to others, provided that you do not + charge any fee for such copies and further provided that + each copy contains this Copyright Notice, whether + distributed in print or electronically. + + + + + + + + + + + + + + diff --git a/docs/src/site/docbook/reference/infrastructure.xml b/docs/src/site/docbook/reference/infrastructure.xml new file mode 100644 index 000000000..9778b9768 --- /dev/null +++ b/docs/src/site/docbook/reference/infrastructure.xml @@ -0,0 +1,529 @@ + + + + The Spring Batch Infrastructure + +
+ Introduction to the + Spring Batch Infrastructure + + Spring Batch is a Pipe and Filters architecture. The Spring Batch + Infrastructure implements key services that enable a high volume of + throughput. These include: + + Input and Output Resource Faciltities - the management of + input items including reading, validating and mapping raw input to + objects. + + + + Item Providers and Processors - strategy interfaces for + providing and processing the data for a given batch stage + execution. + + + + Validation- interface to support pluggable validation + strategies to ensure the integrity of the input source items or, in + other words, object level validation. + + + + RepeatTemplates- the Repeat Template is responsible for + repeatedly invoking an operation on the Input Provider pulling input + items from an input source until there are no more items to be + processed. + + + + RetryTemplates - a mechanism for attempting to reprocess an + input item that has thrown an exception. + + + + Support for Statistics - an interface that dependent projects + can use to implement application specific statistics. + + + + Transaction semantics for batch - support facilities for + giving transaction extensions used by the batch architecture. + + + + + + + + + + + + + Figure 1: Spring Batch Pipe and Filter + Design + + + The Batch Lifecycle is simple. Data comes in one side of the pipe. + It is then parsed, validated and transformed and handed off for business + logic processing. That processing can be as simple as loading records into + a database or as complicated as supporting batch job styles of generating + reports, conversion, pdf generation, generation of high volume print + formats, etc. Spring Batch provides a framework for simplifying the + handling of input and output resources so that developers can concentrate + on what needs to happen during the processing steps. +
+ +
+ Input and Output Sources + + + + Input Source - This is an interface responsible for reading + records from an input stream and also possibly for mapping these + records to objects. It is the responsibility of the implementing + class to decide which technology to use for mapping and configuring + an input source. + + + + Ouput Source - this is the interface for the generation output + operations and works conversely from the input source for + serializing processed data to the appropriate targeted output + source, which includes files, databases or queues. + + + + The Input Source is a basic interface for generic input operations. + Subclasses implementing this interface will be responsible for reading + records from input stream and also possibly for mapping these records to + objects. Generally it is the responsibility of implementing class to + decide which technology to use for mapping and how it should be + configured. A picture of the I/O hierarhcy is helpful in understanding + their place within the spring batch infrastructure. + + + + + + + + + + + Figure 2: Input/Output Sources + + + A description of input and output resources that spring batch + supports are the following: + + + + File - File Sources read and write lines of data from a flat + file that typically describe records with fields of data defined by + fixed positions in the file or delimited by some special character + (e.g. a comma). There is a line tokenizer associated with input + sources and a line aggregator associated with the output + source. + + + + SQL - a database resource accessed that returns resultsets + that can be mapped to objects for processing. The default SQL Input + Sources invoke a RowMapper to return objects, keep track of the + current row if restart is required, basic statistics, and some + transaction enhancements that will be explained later. + + Note: There is no SQL Output Source because there is no state + to manage whereas the SQL Input Source requires state to monitor + skips, the current position in the input source, restart data, + etc. + + + + XML - an XML input and output sources process XML + independently of technologies used for parsing, mapping and + validating objects. Input data allows for the validation of and XML + file against and XSD schema. The input template provides for + restart, skip, statistics and transaction features by implementing + the corresponding interfaces. + + + +
+ + + Flat File Sources + + + + Flat File Input Sources - Flat File Input Sources are basic input + sources that read data from a file and return it as structured tuples in + the form of FieldSet instances. Flat File Input Sources are further + refined as both fixed length and delimited formats. + + + + + + + + + + + + + + + + + + + Figure 1: Flat File Input Source Collaborations + + + + + + + The location of the file is defined by the resource property. + There are only a few methods exposed through a resource service. A + resource is used to help locate, open, and close resources. It can be as + simple as: + Resource resource = new FileSystemResource("resources/trades.csv"); + + + + + In complex batch environments the directory structures are often + managed by the EAI infrastructure where drop zones for external + interfaces are established for moving files from ftp locations to batch + processing locations and vice versa. File moving utilities are beyond + the scope of the batch architecture but its not unusual for batch job + streams to include file moving utilities as steps in the job stream. + It's sufficient to know that the batch architecture only needs to know + how to find the files to be processed and it begins the process of + feeding the data into the pipe from this starting point. + + + + To separate the structure of the file, LineTokenizer, or one of it + subclasses, is used to parse data obtained from the file. Flat File + Input Sources, as mentioned above, typically come in two forms, fixed + and delimited. A fixed length input record is where the fields are + assigned fixed locations within a line of a file. An example would be: + + 12345678901234567890123456789012345678901234567890 + AbduKa00Abdul-Jabbar Karim rb19741996 + AbduRa00Abdullah Rabih rb19751999 + AberWa00Abercrombie Walter rb19591982 + AbraDa00Abramowicz Danny wr19451967 + AdamBo00Adams Bob te19461969 + AdamCh00Adams Charlie wr19792003 + + + On the other hand a delimited record format might look like the following: + + + + AbduKa00,Abdul-Jabbar,Karim,rb,1974,1996 + AbduRa00,Abdullah,Rabih,rb,1975,1999 + AberWa00,Abercrombie,Walter,rb,1959,1982 + AbraDa00,Abramowicz,Danny,wr,1945,1967 + AdamBo00,Adams,Bob,te,1946,1969 + AdamCh00,Adams,Charlie,wr,1979,2003 + + + + + + Neither of these formats are particularly self describing but are + still very much in use in flat file exchanges between system interfaces. + Both formats share in common the requirement to read in a line of data + (a String) and parse it into tokens that can be mapped to an object (or + objects) to be passed to the ItemProcessor. As you can see, there are + two required dependencies of the input source; the first is a resource + to read in, which is the file to process. The second dependency is a + LineTokenizer, which will be discused below. + + +
+ +
+ Configuring and using FieldSets + + A FieldSet is Spring Batch’s abstraction for typing fields from a + flat file data source. It allows developers to work with file input in + much the same way as they would work with database input. A FieldSet is + conceptually very similar to a Jdbc Result Set. FieldSets only require + one argument, a list of tokens. Optionally you can also configure in the + names of the fields so that the fields may be accessed either by index + or name as patterned after the JdbcResultSet. In code it means it's as + simple as: + + + tokens = new String[] { "TestString", "true", "C", "10", "-472", "354224", "543", "124.3", "424.3", "324", + null, "2007-10-12", "12-10-2007", "" }; + names = new String[] { "String", "Boolean", "Char", "Byte", "Short", "Integer", "Long", "Float", "Double", + "BigDecimal", "Null", "Date", "DatePattern", "BlankInput" }; + + fieldSet = new FieldSet(tokens, names); + assertTrue(fieldSet.getFieldCount() == 14); + +
+ +
+ Configuring and Using + LineTokenizers + + The interface for a LineTokenizer is very simple, given a string; + it will return a FieldSet that wraps the results from tokenizing the + provided string. The tokens are created through a + LineTokenizer and a + FieldSetMapper is used to map a the + FieldSet to an object. The framework provides a few + convenience classes, the FieldSetInputSource and + the SimpleFlatFileInputSource. They provide a + convenient way to read the FieldSet. The FieldSet + is configured as a property for an Item Provider, which wraps an Input + Source. You can see this in the following example: + + + <property name="itemProvider"> + <bean class="org.springframework.batch.sample.item.provider.PlayerItemProvider"> + <property name="inputSource" ref="playerFileInputSource" /> + <property name="fieldSetMapper"> + <bean class="org.springframework.batch.sample.mapping.PlayerMapper" /> + </property> + </bean> + </property> + + + The LineTokenizer is just one additional property to the + InputSource as seen here: + + + <property name="tokenizer"> + <bean + class="org.springframework.batch.io.file.support.transform.DelimitedLineTokenizer"> + <property name="names" + value="ID,lastName,firstName,position,birthYear,debutYear" /> + </bean> + </property> + + + And, as you can see, the field names will get passed in the the + mapper. The actual mapping provided by the developer would then look as + simple as: + + +public class PlayerMapper implements FieldSetMapper { + public Object mapLine(FieldSet fs) { + if(fs == null){ + return null; + } + + Player player = new player(); + player.setID(fs.readString("ID")); + player.setLastName(fs.readString("lastName")); + player.setFirstName(fs.readString("firstName")); + player.setPosition(fs.readString("position")); + player.setDebutYear(fs.readInt("debutYear")); + player.setBirthYear(fs.readInt("birthYear")); + + return player; + } +} + +
+ +
+ Output Sources + + The output source is similar in functionality to the input source + with the exception that the operations are reversed. They still need to + be located, opened and closed but they differ in the case that we write + to output sources. In the case of databases or queues these may be + inserts, updates or sends. The format of the serialization of the output + source is specific for every batch job. +
+
+ +
+ SQL Sources + + SQL input sources can be configured for various reasons, for + example: + + + + a staging table for large volumes of sorted data that was loaded + from flat files + + + + the beginning of an outbound collection of data targeted for an + external flat file interface + + + + the target of a triggered event like "collect all cases that can + be automatically closed" + + Spring Batch supports two approaches for accessing a SQL Input + Source; 1) a cursor driven input source and 2) an indexed based Input + Query. The cursor driven input source is named because it utilizes a + jdbc cursor to stream over the SQL input source whereas an indexed + based input query is designed for easy division of the input into + ranges. + + + + +
+ +
+ XML Input and Output + + Spring Batch provides transactional infrastructure for both reading + XML records and mapping them to Java objects as well as writing Java + objects as XML records. + + StAX API is used for I/O as other standard XML APIs do not fit batch + processing requirements (DOM loads the whole input into memory at once and + SAX controls the parsing process allowing the user only to provide + callbacks). + + Spring Batch is not tied to any particular OXM technology. Typical + use is to delegate OXM to Spring WS which provides uniform abstraction for + the most popular OXM technologies. However dependency on Spring WS is + optional and you can choose to implement Spring Batch specific interfaces + if desired. + + Lets take a closer look how XML input and output work in batch. It + is assumed the XML resource is a collection of 'fragments' corresponding + to individual records. Note that OXM tools are designed to work with + standalone XML documents rather than XML fragments cut out of an XML + document, therefore the Spring Batch infrastructure needs to work around + this fact (as described below). + + On input the reader reads the XML resource until it recognizes a new + fragment is about to start (by matching the tag name by default). The + reader creates a standalone XML document from the fragment (or at least + makes it appear so) and passes the document to a deserializer (typically a + wrapper around Spring WS Unmarshaller) to map the XML to a Java + object. + + Output works symetrically to input. Java object is passed to a + serializer (typically a wrapper around Spring WS Marshaller) which writes + to output using a custom event writer that filters the StartDocument and + EndDocument events produced for each fragment by the OXM tools. + + For example configuration of XML input and output see the sample + xmlStaxJob. //TODO inline the example once it is not subject to change + + show sample input file + + +
+ +
+ Item Providers and Processors + + We finally arrive at the Item Provider, We've already alluded to + Item Providers in some of the code samples above. +
+ +
+ Validating Input + + +
+ +
+ Repeat Templates + + One of the most fundamental concepts in the batch architecture is + the Repeat Template. The Repeat Template is responsible for repeatedly + invoking an operation on the Input Provider pulling input items from an + input source until there are no more items to be processed. One + interesting analogy used by Dierk Koenig in the book "Groovy in Action" is + a boiler vs. a continuous-flow heater. In this analogy he illustrates how + XML parsers can typically be divided into those that read the entire input + before process begins like DOM Parsers vs. those that stream over the + input like SAX parsers. Spring Batch is a continuous-flow heater and uses + the RepeatTemplate as the mechanism to keep the hot water or input stream + in constant flow. + + Many times batch processes are not only working on non-transaction + input sources like files but the output is a transactional resource such + as a queue or database. A common scenario when a batch job is a datastream + coming from a flat file interface is to have a file or files as input + sources and a database resource as the output source. In this case the + repeat templates can be used like the following: + + + + + + + + + Figure 2: Simple Batch Pseduo code for Repeat + Templates + + + In this batch scenario an outer RepeatTemplate initialies the + continuous flow, a TransactionTemplate wraps the input and output + resources and an inner RepeatTemplate manages the commit interval or + chunks of data to be processed. The Business Logic occurs in the input and + output of single items. Of course this is a simplistic view of how batch + really works. Input can be quite complex with multiple files and + complicated validation scenarios. Conversely, the output source can also + be quite complex in determining how the records will be stored in the + database. Spring Batch makes no assumptions about how simple or complex + the business processing is within the RepeatTemplates. It's only job is to + keep the flow moving from the Item Provider to the Item Processor as + quickly as possible. The repeat template can process records irrespective + of the batch architecture. A simple example would be: + + RepeatTemplate template = new RepeatTemplate(); + Resource resource = new FileSystemResource("resources/trades.csv"); + TradeProcessor executor = new TradeProcessor(); + TradeItemProvider provider = null; + try { + provider = new TradeItemProvider(resource); + } catch (Exception e) { + // TODO Auto-generated catch block + e.printStackTrace(); + } + template.iterate(new ItemProviderRepeatCallback(provider, executor)); + + + A RepeatTemplate has an exception policy that can be + leveraged +
+ +
+ Retry Template + + The retry template is used as a way to overcome failures in the + stream +
+
\ No newline at end of file diff --git a/docs/src/site/docbook/reference/namespace.xml b/docs/src/site/docbook/reference/namespace.xml new file mode 100644 index 000000000..176bc0f1f --- /dev/null +++ b/docs/src/site/docbook/reference/namespace.xml @@ -0,0 +1,11 @@ + + + Namespace Support + +
+ Namespace Support + Document Namespace support here. +
+ +
diff --git a/docs/src/site/docbook/reference/outline.xml b/docs/src/site/docbook/reference/outline.xml new file mode 100644 index 000000000..0a319f87b --- /dev/null +++ b/docs/src/site/docbook/reference/outline.xml @@ -0,0 +1,62 @@ + + + + + The Spring Batch - Reference Documentation + Wayne Lund, Waseem Malik, Lucas Ward, Scott Wintermute, + Kerry O'Brien, Tomi Vanek + May 2007 + + + +
+ Chapter 1: Spring Batch Introduction + Overview of the Spring Batch Architecture - the Spring Batch Reference Model +
+ +
+ <ulink url="infrastructure.html">Chapter 2: The Spring Batch Infrastructure</ulink> + Infrastructure covers Repeat Template, I/O facilities and the RetryTemplate +
+ +
+ <ulink url="core.html">Chapter 3: Spring Batch Core</ulink> + Describe the domain language of batch and how the pieces fit together. +
+ +
+ <ulink url="execution.html">Chapter 4: Spring Batch Execution</ulink> + Describe the simple batch execution environment. +
+ +
+ <ulink url="application.html">Chapter 5: Spring Batch Applications</ulink> + Describe the solution space for spring batch. +
+ +
+ <ulink url="samples.html">Chapter 6: Spring Batch Samples</ulink> + The documentation for samples goes here +
+ +
+ <ulink url="batch-job-testing.html">Chapter 7: Unit and Integration Testing Batch Jobs</ulink> + The documentation for unit and integration testing of batch jobs goes here. +
+ +
+ <ulink url="batch-performance-testing.html">Chapter 8: Performance Testing Batch Jobs</ulink> + How to performance test batch jobs. +
+ +
+ <ulink url="glossary.html">Chapter 9: Glossary</ulink> + (Should this be Appendix A?) The Batch Glossary documents common terms used in the batch processing domain. +
+ +
+ More sections that may come later? + More advanced info about building Spring Batch, how to contribute, JMS integration, management with JMX, the Spring Batch data model, integration with schedulers (eg quartz) +
+
+ diff --git a/docs/src/site/docbook/reference/partitioned-containers.xml b/docs/src/site/docbook/reference/partitioned-containers.xml new file mode 100644 index 000000000..7f9599cc0 --- /dev/null +++ b/docs/src/site/docbook/reference/partitioned-containers.xml @@ -0,0 +1,10 @@ + + + Batch IO Support +
+ Partitioned Containers + Much stuff goes here. +
+ +
diff --git a/docs/src/site/docbook/reference/samples.xml b/docs/src/site/docbook/reference/samples.xml new file mode 100644 index 000000000..e02921da3 --- /dev/null +++ b/docs/src/site/docbook/reference/samples.xml @@ -0,0 +1,1094 @@ + + + + Sample Jobs + +
+ Overview of Batch Samples + + There is considerable variability in the types of input and output + formats in batch jobs. There is also a number of options to consider in + terms of how the types of strategies that will be used to handle skips, + recovery, and statistics. However, when approaching a new batch job there + are a few standard questions to answer to help determine how the job will + be written and how to utilize the services offered by the spring batch + framework. Consider the following: + + + + How do I configure this batch job? In the reference applications + the pattern is to follow the convention of nameOf + Job.xml. Each sample will identify the XML definition used to + configure the job. Job configurations that leverage a common execution + environment have many common items in their respective + configurations. + + + + What is the input source? Each sample batch job will identify + its input source. + + + + What is my output source? Each sample batch job will identify + its output source. + + + + How are records read and validated from the input source? This + refers to the input type and its format (e.g. flat file with fixed + position, comma separated or XML, etc.) + + + + What is the policy of the job if a input record fails the + validation step? The most important aspect is whether the record can + be skipped so that processing can be continued. + + + + How will I process the data and write to the output source? How + and what business logic is being applied to the processing of a + record. + + + + How do I recover from an exception while operating on the output + source? There are numerous recovery strategies that can be applied to + handling errors on transactional targets. The reference applications + will provide a feeling for some of the choices. + + + + Can I restart the job and if so which strategy will I use to + restart the job? The reference applications will show some of the + options available to jobs and what the decision criteria is for the + respective choices. + + + + Samples + + + Reference Applications Table of Features + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + Job / Feature + + delimited input + + fixed-length input + + xml input + + db driving query input + + db cursor input + + delimited output + + fixed-length output + + db output + + skip + + restart + + quartz scheduling + + + + simpleTaskletJob + + + + + + + + + + + + + + + + + + + + + + + + + + fixedLengthImport + + + + + + + + + + + + + + + + + + + + + + + + + + multi-line order + + + + + + + + + + + + + + + + + + + + + + + + + + quartzBatch + + + + + + + + + + + + + + + + + + + + + + + + + + simple skip sample + + + + + + + + + + + + + + + + + + + + + + + + + + Skip And Restart Sample + + + + + + + + + + + + + + + + + + + + + + + + + + SQL Cursor Trade Job + + + + + + + + + + + + + + + + + + + + + + + + + + Trade Job + + + + + + + + + + + + + + + + + + + + + + + + + + XML Job + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+ +
+ <anchor id="a.common-sample-setup" /> Common Sample Test + Structures + + The sample applications, although executing within unit test + frameworks are actually integration tests that configure the simple batch + execution environment, the job with its respective steps and tasklets, and + wire in the infrastructure services used by the job. On a quick inspection + the sample jobs, especially if new to Spring, appear to use a lot of auto + magic. However, this is simply the power of Spring to help configure + applications and allow developers to focus on the application and not + infrastructure. The common test structures are the following; + + + + All test cases subclass AbstractLifecycleSpringContextTests: + This class extends a convenience class provided by Spring, + AbstractDependencyInjectionSpringContextTests, that is populated by + Dependency Injection. This effectively serves as a substitute Batch + Launcher that conveniently describes the location of the application + context and loads the beans enlisted by the job into the application + context. In addition, once the batch execution environment has + completed the wiring of the application context, the Batch Bootstrap + process launches the BatchLauncher via its start() method. + + + + All test cases have a common configuration structure: If using + Spring IDE you will see the following configuration hierarchy: + + + + The Job Specific Configuration + + + + The Simple Batch Execution Environment Definition + + + + The Data Source Context + + + + +
+ +
+ <anchor id="a.simple-tasklet" /> Simple Tasklet + Job + + The goal is to show the simplest use of the batch framework with a + single job with a single step&nbsp;where the tasklet + processes&nbsp;one input source to one output source. + + Description: This job is defined by + simpleTaskletJob.xml file. Job itself is defined by element + simpleTaskletJob. Each job consists of several steps, these steps are + defined in steps property. In this example we have only one step. The step + defines a tasklet that is responsible for processing trades. In this case + processing will be handled by SimpleTradeTasklet class. Each tasklet must + implement execute() method. All processing of business data should be + handled by this method. In this example the tradeFieldSetMapper obtains + the data from the input source and maps the line to the Trade object. + + + trade = (Trade) tradeFieldSetMapper.mapLine(inputSource.readFieldSet()); + + If data exists and an object is returned it is simply passed to + the output source. If there is no data to read an ExitStatus with the + status of FINISHED is returned from the Tasklet. + + Method read() gets the data from + the input template defined and maps it to an object using mapper defined + in XML definition. This sample uses FlatFileInputTemplate class as input + template. This template reads the whole line from the file and pass it to + tokenizer which knows the structure of the line. Location of the file is + defined by fileLocatorStrategy property, structurte of the line is defined + by fixedFileDescriptor. Result of parsing the line is stored in FieldSet, + which is used by mapper to create value object. In our example we use + DefaultLineMapper which creates an instance of Trade class. + + Method process() is quite simple - + just writes trade object using DbTradeWriter class. This class writes + values obtained from an object to the database. + + Specific information: This job has + whole logic implemented in Tasklet. It is not using Data provider as well + as Tasklet processor, which is typical way how to handle data. + + XML definition: + simpleTaskletJob.xml + + [Note: we need to document Spring IDE in setup and installation so + we can use to describe the project. Also, if we could also publish we can + provide links to the graphics from docs. This is a sample only]. + + Visualization of the spring configuration through Spring-IDE exposes + the structure of a job configuration. The following is the visualization + of the Simple Tasklet Job configuration. See Spring IDE . + + + + + + + + + + +
+ +
+ Spring IDE Graph of Simple Tasklet Job + Configuration. + + + + Figure: + + + Simple Tasklet Job Configuration + + For simplicity we are only displaying the job configuration + itself and leaving out the details of the supporting batch execution + environment configuration. The source view of the configuration is + as follows: + + + + + + <import resource="BatchArchConfig.xml" /> + <bean id="simpleTaskletJob" parent="Job"> + <property name="name" value="fixedLengthImportJob" /> + <property name="steps"> + <list> + <bean id="tradeStep" parent="Step"> + <property name="name" value="ImportTradeDataStep" /> + <property name="module"> + <bean class="com.accenture.adsj.refapp.batch.module.SimpleTradeTasklet"> + <property name="inputTemplate" ref="fileInputTemplate" /> + <property name="tradeDbWriter" ref="tradeWriter" /> + </bean> + </property> + <property name="commitFrequency" value="5" /> + <property name="startPolicy"> + <bean class="org.springframework.batch.container.conf.StartPolicy"> + <property name="ignoreComplete" value="true" /> + <property name="restartEnabled" value="true" /> + <property name="startlimit" value="12" /> + </bean> + </property> + <property name="exceptionPolicy"> + <bean class="org.springframework.batch.container.conf.ExceptionPolicy"> + <property name="totalExceptionLimit" value="20" /> + <property name="transactionInvalidExceptionLimit" value="20" /> + <property name="transactionValidExceptionLimit" value="5" /> + </bean> + </property> + </bean> + </list> + </property> + </bean> + <bean id="tradeWriter" class="com.accenture.adsj.refapp.batch.dao.DbTradeWriter"> + <property name="jdbcTemplate" ref="jdbcTemplate" /> + <property name="incrementer"> + <bean parent="incrementerParent"> + <property name="incrementerName" value="TRADE_SEQ" /> + </bean> + </property> + </bean> + <bean id="fileInputTemplate" class="org.springframework.batch.container.io.file.support.FlatFileInputTemplate"> + <property name="name" value="FileInputSource" /> + <property name="fileLocatorStrategy" ref="fileLocator" /> + <property name="tokenizer"> + <bean class="org.springframework.batch.container.io.file.support.FixedLineTokenizer"> + <property name="fileDescriptor" ref="fixedFileDescriptor" /> + </bean> + </property> + </bean> + +<bean id="fixedFileDescriptor" class="org.springframework.batch.container.io.support.DefaultFileDescriptor"> + <property name="recordDescriptors"> + <bean class="org.springframework.batch.container.io.support.DefaultRecordDescriptor"> + <property name="fieldDescriptors"> + <list> + <bean class="org.springframework.batch.container.io.support.DefaultFieldDescriptor"> + <property name="name" value="ISIN" /> + <property name="length" value="12" /> + </bean> + <bean class="org.springframework.batch.container.io.support.DefaultFieldDescriptor"> + <property name="name" value="Quantity" /> + <property name="length" value="3" /> + </bean> + <bean class="org.springframework.batch.container.io.support.DefaultFieldDescriptor"> + <property name="name" value="Price" /> + <property name="length" value="5" /> + </bean> + <bean class="org.springframework.batch.container.io.support.DefaultFieldDescriptor"> + <property name="name" value="Customer" /> + <property name="length" value="9" /> + </bean> + </list> + </property> + </bean> + </property> +</bean> +<bean id="tradeLineMapper" class="com.accenture.adsj.refapp.batch.mapping.TradeRowMapper" /> +<bean class="com.accenture.adsj.refapp.batch.advice.LogAdvice" id="logAdvice" /> +<aop:config> + <aop:aspect id="logging" ref="logAdvice"> + <aop:around pointcut-ref="pointcut" method="doBasicLogging" /> + <aop:pointcut id="pointcut" expression="execution(* org.springframework.batch.container.dao.*.*(..))" /> + </aop:aspect> +</aop:config> + </beans> + + + + You should take the time to make sure you understand the + relationship of the xml configuration with the visualization as provided + by Spring IDE. [Note: this will be updated when we use the namespace + handler]. + + Input source: file with fixed row + structure + + In this example we are using a simple fixed length record structure + that can be found in the project at + REFAPP_INSTALL_HOME/testBatchRoot/job_data/simpleTaskletJob/input/20070122.teststream.ImportTradeDataStep.txt. + There's generally a considerable amount of thought that goes into + architecting the folder structures for batch file management. See [provide + a link to DefaultFileStrategy]. The only point to note here is the + ImportTradeDataStep matches the name of the step in the configuration and + the fixed length records look like: + + 20070122.teststream.ImportTradeDataStep.txt + + UK21341EAH4597898.34customer1 + UK21341EAH4611218.12customer2 + UK21341EAH4724512.78customer2 + UK21341EAH48108109.25customer3 + UK21341EAH49854123.39customer4 + + Looking back to the configuration file you will see where this is + documented in the propery of the DefaultRecordDescriptor. You can see the + following: + + + + + + + + + + FieldName + + Length + + + + ISIN + + 12 + + + + Quantity + + 3 + + + + Price + + 5 + + + + Customer + + 9 + + + + + + Output target: database + + Data Provider: data provider is not + used, all functionality is implemented directly in Tasklet. + + Tasklet processor: module processor + is not used, all functionality is implemented directly in Tasklet. +
+ +
+ Fixed Length Import Job + + The goal is to demonstrate a typical scenarion of importing data + from a fixed-length file to database + + Description: This job shows a more + typical scenario, when reading input data and processing the data is + cleanly separated. The data provider is responsible for reading input and + mapping each record to a domain object, which is then passed to the module + processor. The module processor handles the processing of the domain + objects, in this case it only writes them to database. + + XML definition: + fixedLengthImportJob.xml + + Input source: file with fixed row + structure + + Output target: database + + Data Provider: + DefaultFlatFileDataProvider which uses the injected FlatFileInputTemplate + to read input and the DefaultLineMapper to map each line to an object + according to the file descriptor. + + Tasklet processor: module processor + does not do any special processing, it just writes the data to database + using a DAO object (called OutputSource in this case, because it is + specialized for writing to database, it has no methods for reading + data). +
+ +
+ Multiline Order Job + + The goal is to demostrate how to handle a more complex file input + format, where a record meant for processing inludes nested records and + spans multiple lines + + XML definition: + multilineOrderJob.xml + + Input source: file with multiline + records + + Output target: file with multiline + records + + Data Provider: OrderDataProvider is + an example of a non-default programmatic data provider. It reads input + until it detects that the multiline record has finished and encapsulates + the record in a single domain object. + + Tasklet processor: module processor + passes the object to a an injected 'report service' which in this case + writes the output to a file do demonstrate how to use the + FlatFileOutputTemplate for writing multiline output according to a file + descriptor. +
+ +
+ Quartz Batch + + The goal is to demonstrate how to schedule job execution using + Quartz scheduler + + XML definition: + quartzBatch.xml + + Description: First, declares + launcher beans. Each launcher bean is able to launch a job using injected + arguments. Second, triggers are declared saying when the launchers should + be run. Last, there is the scheduler bean, where the triggers are + registered. +
+ +
+ Simple Skip Sample + + Document how skip works. +
+ +
+ Skip And Restart Sample + + Document how Skip and Restart Sample Works +
+ +
+ SQL Cursor Trade Job + + Document how SQL Cursor Trade Job works +
+ +
+ Trade Job + + The goal is to show a reasonably complex scenario, that would + resemble the real-life usage of the framework. + + Description: This job has 3 steps. + First, data about trades is imported from a file to database. Second, the + data about trades is read from the database and credit on customer + accounts is decreased appropriately. Last, a report about customers is + exported to a file. + + XML definition: tradeJob.xml - the + job definition, tradeJobIo.xml - input and output configuration, + tradeJobAop.xml - optional AOP logging + + Description: This job has 3 steps. + First, data about trades is imported from a file to database. Second, the + data about trades is read from the database and credit on customer + accounts is decreased appropriately. Last, a report about customers is + exported to a file. +
+ +
+ XML Job + + Document how the sample XML job works +
+ +
+ Football Job + + The final Job is an Football statistics loading job. We’ll give it + the id of “footballjob” in our configuration file. Before diving into the + batch job, we’ll examine the two input files that need to be loaded. First + is ‘player.csv’, which can be found in the samples project under + src/main/resources/data/footballjob/input/. Each line within this file + represents a player, with a unique id, the player’s name, position, etc: + + AbduKa00,Abdul-Jabbar,Karim,rb,1974,1996 + AbduRa00,Abdullah,Rabih,rb,1975,1999 + AberWa00,Abercrombie,Walter,rb,1959,1982 + AbraDa00,Abramowicz,Danny,wr,1945,1967 + AdamBo00,Adams,Bob,te,1946,1969 + AdamCh00,Adams,Charlie,wr,1979,2003 + + + One of the first noticeable characteristics of the file is that each + data element is separated by a comma, a format most are familiar with + known as ‘CSV’. Other separators such as pipes or semicolons could just as + easily be used to delineate between unique elements. In general, it falls + into one of two types of flat file formats: delimited or fixed length. + Because both input files in this example are comma delimited, we’ll skip + over fixed length for now, other than to say that the only difference + between the two types is that fixed length formatting determines the + separation between elements by assigning each element a ‘fixed length’ in + which to reside, rather than using a character that hopefully doesn’t + exist in the data itself to separate individual elements. + + The second file, ‘games.csv’ is formatted the same as the previous + example, and resides in the same directory: + AbduKa00,1996,mia,10,nwe,0,0,0,0,0,29,104,,16,2 + AbduKa00,1996,mia,11,clt,0,0,0,0,0,18,70,,11,2 + AbduKa00,1996,mia,12,oti,0,0,0,0,0,18,59,,0,0 + AbduKa00,1996,mia,13,pit,0,0,0,0,0,16,57,,0,0 + AbduKa00,1996,mia,14,rai,0,0,0,0,0,18,39,,7,0 + AbduKa00,1996,mia,15,nyg,0,0,0,0,0,17,96,,14,0 + + + Each line in the file represents an individual player’s performance + in a particular game, containing such statistics as passing yards, + receptions, rushes, and total touchdowns. + + Our example batch job is going to load both files into a database, + and then combine each to summarize how each player performed for a + particular year. Although this example is fairly trivial, it shows + multiple types of input, and the general style is a common batch scenario. + That is, summarizing a very large dataset so that it can be more easily + manipulated or viewed by an online web-based application. In an enterprise + solution the third step, the reporting step, could be implemented through + the use of Eclipse BIRT or one of the many Java Reporting Engines. Given + this description, we can then easily divide our batch job up into 3 + ‘steps’: one to load the player data, one to load the game data, and one + to produce a summary report: + + NOTE:One of the nice features of Spring is a project called Spring + IDE. When you download the project you can install Spring IDE and add the + Spring configurations to the IDE project. This is not a tutorial on Spring + IDE but the visual view into Spring beans is helpful in understanding the + structure of a Job Configuration. Spring IDE produces the following + diagram: + + + + + + + + + + + + + + + Figure 3 - Spring Bean Job Configuration + + This corresponds exactly with the footballJob.xml job configuration + file which can be found in the jobs folder under src/main/resources. When + you drill down into the footballjob you will see that the configuration + has a list of steps: + <property name="steps"> + <list> + <bean id="playerload"> ... </bean> + <bean id="gameLoad"> ... </bean + <bean id="playerSummarization"> ... </bean> + </list> + </property> + + + The step is run until there is no more input to process, which in + this case would mean that each file has been completely processed. To + describe it in a more narrative form: The first step, playerLoad, begins + executing by grabbing one line of input from the file, and parsing it into + a domain object. That domain object is then passed to a dao, which writes + it out to the PLAYERS table. This action is repeated until there are no + more lines in the file, causing the playerLoad step to finish. Next, the + gameLoad step does the same for the games input file, inserting into the + GAMES table. Once finished, the playerSummarization step can begin. Unlike + the first two steps, playerSummarization’s input comes from the database, + using a Sql statement to combine the GAMES and PLAYERS table. Each + returned row is packaged into a domain object and written out to the + PLAYER_SUMMARY table. + + Now that we’ve discussed the entire flow of the batch job, we can + dive deeper into the first step: playerLoad: + <bean id="playerload" class="org.springframework.batch...SimpleStepConfiguration"> + <property name="commitInterval" value="100" /> + <property name="tasklet"> + <bean class="org.springframework...RestartableItemProviderTasklet"> + <property name="itemProvider">...</property> + <property name="itemProcessor">...</property> + </bean> + </property> + </bean> + + + The root bean in this case is a StepConfiguration, which can be + considered a ‘blueprint’ of sorts that tells the execution environment + basic details about how the batch job should be executed. It contains two + properties: (others have been removed for greater clarity) commitInterval + and tasklet. The Tasklet is the main abstraction representing the + developer’s business logic within the batch job. After performing all + necessary startup, the framework will periodically delegate to the + Tasklet. In this way, the developer can remain solely concerned with their + business logic. In this case, the Tasklet has been split into two classes: + + + Item Provider – the item provider is the + source of the information pipe. At the most basic level input is + read in from an input source, parsed into a domain object and + returned. In this way, the good batch architecture practice of + ensuring all data has been read before beginning processing can be + enforced, along with providing a possible avenue for reuse. + + + + Item Processorr – this is the business + logic. At a high level, the ItemProcessor takes the item returned + from the ItemProvider and ‘processes’ it. In our case it’s a data + access object that is simply responsible for inserting a record into + the PLAYERS table. As you can see the developer does very + little. + + + + Clearly, the developer does very litt. Simply provide a job + configuration with a configured number of steps, an Item Provider + associated to some type of input source, and Item Processor associated to + some type of output source and a little mapping of data from flat records + to objects and the pipe is ready wired for processing. + + The other property to the StepConfiguration, commitInterval, gives + the framework vital information about how to control transactions during + the batch run. Due to the large amount of data involved in batch + processing, it is often advantageous to ‘batch’ together multiple Logical + Units of Work into one transaction, since starting and committing a + transaction is extremely expensive. For example, in the playerLoad step, + the framework calls the execute() method on the Tasklet, which then calls + next() on the ItemProvider. The ItemProvider reads one record from the + file, then returns a domain object representation which is passed to the + processor. The processor then writes the one record to the database. It + can then be said that one iteration = one call to Tasklet.execute() = one + line of the file. Therefore, setting your commitInterval to 5 would result + in the framework committing a transaction after 5 lines have been read + from the file, with 5 resultant entries in the PLAYERS table. + + Following the general flow of the batch job, the next step is to + describe how each line of the file will be parsed from its string + representation into a domain object. The first thing the provider will + need is an InputSource, which is provided as part of the Spring Batch + infrastructure. Because the input is flat-file based, a + FlatFileInputSource is used: +<bean id="playerFileInputSource" +class="org.springframework.batch.io.file.support.DefaultFlatFileInputSource"> + <property name="resource"> + <bean class="org.springframework.core.io.ClassPathResource"> + <constructor-arg value="data/footballjob/input/player.csv" /> + </bean> + </property> + <property name="tokenizer"> + <bean class = "org.springframework.batch.io.file.support.transform.DelimitedLineTokenizer"> + <property name="names" + value="ID,lastName,firstName,position,birthYear,debutYear" /> + </bean> + </property> +</bean> + + + There are two required dependencies of the input source; the first + is a resource to read in, which is the file to process. The second + dependency is a LineTokenizer. The interface for a LineTokenizer is very + simple, given a string; it will return a FieldSet that wraps the results + from splitting the provided string. A FieldSet is Spring Batch’s + abstraction for flat file data. It allows developers to work with file + input in much the same way as they would work with database input. All the + developers need to provide is a FieldSetMapper (similar to a Spring + RowMapper) that will map the provided FieldSet into an Object. Simply by + providing the names of each token to the LineTokenizer, the ItemProvider + can pass the FieldSet into our PlayerMapper, which implements the + FieldSetMapper interface. There is a single method, mapLine(), which maps + FieldSets the same way that developers are comfortable mapping ResultSets + into Java Objects, either by index or fieldname. This behavior is by + intention and design similar to the RowMapper passed into a JdbcTemplate. + You can see this below: +public class PlayerMapper implements FieldSetMapper { + + public Object mapLine(FieldSet fs) { + + if(fs == null){ + return null; + } + + Player player = new Player(); + player.setID(fs.readString("ID")); + player.setLastName(fs.readString("lastName")); + player.setFirstName(fs.readString("firstName")); + player.setPosition(fs.readString("position")); + player.setDebutYear(fs.readInt("debutYear")); + player.setBirthYear(fs.readInt("birthYear")); + + return player; + } +} + + + The flow of the ItemProvider, in this case, starts with a call to + readFieldSet on the InputSource. The next line in the file is read in as a + String and passed into the provided LineTokenizer. The LineTokenizer + splits the line at every comma, and creates a FieldSet using the created + String array and the array of names passed in. (Note: it is only necessary + to provide the names if you wish to access the field by name, rather than + by index). + + Once the domain representation of the data has been returned by the + provider, (i.e. an Player object) it is passed to the ItemProcessor, which + is essentially a Dao that uses a Spring JdbcTemplate to insert a new row + in the PLAYERS table. + + The next step, gameLoad, works almost exactly the same as the + playerLoad step, except the games file is used. + + The final step, playerSummarization, is much like the previous two + steps, it is split into a provider that reads from an InputSource and + returns a domain object to the processor. However, in this case, the input + source is the database, not a file: +<bean id="playerSummarizationSource" + class="org.springframework.batch.io.sql.SqlCursorInputSource"> + <property name="dataSource" ref="dataSource" /> + <property name="mapper"> + <bean class="sample.mapping.PlayerSummaryMapper" /> + </property> + <property name="sql"> + <value> + SELECT games.player_id, games.year, SUM(COMPLETES), + SUM(ATTEMPTS), SUM(PASSING_YARDS), SUM(PASSING_TD), + SUM(INTERCEPTIONS), SUM(RUSHES), SUM(RUSH_YARDS), + SUM(RECEPTIONS), SUM(RECEPTIONS_YARDS), SUM(TOTAL_TD) + from games, players where players.player_id = + games.player_id group by games.player_id, games.year + </value> + </property> +</bean> + + + The SqlCursorInputSource has three dependences: + + A DataSource + + + + The SqlRowMapper to use for each row. + + + + The Sql statement used to create the Cursor. + + + + When the step is first started, a query will be run against the + database to open a cursor, and each call to inputSource.read() will move + the ‘cursor’ to the next row, using the provided RowMapper to return the + correct object. As with the previous two steps, each record returned by + the provider will be written out to the database in the PLAYER_SUMMARY + table. Finally to run this sample application you can execute the JUnit + test “FootballJobFunctionalTests”, and you’ll see an output showing each + of the records as they are processed. Please keep in mind that AoP is used + to wrap the ItemProcessors and output each record as it is processed to + the logger, which will greatly impact performance. +
+
\ No newline at end of file diff --git a/docs/src/site/docbook/reference/spring-batch-football-graph.JPG b/docs/src/site/docbook/reference/spring-batch-football-graph.JPG new file mode 100644 index 000000000..d8898c135 Binary files /dev/null and b/docs/src/site/docbook/reference/spring-batch-football-graph.JPG differ diff --git a/docs/src/site/docbook/reference/spring-batch-intro.xml b/docs/src/site/docbook/reference/spring-batch-intro.xml new file mode 100644 index 000000000..328ed883e --- /dev/null +++ b/docs/src/site/docbook/reference/spring-batch-intro.xml @@ -0,0 +1,301 @@ + + + + Spring Batch Introduction + +
+ Introduction + + Many applications within the enterprise domain require bulk + processing to perform business operations in mission critical + environments. These business operations include automated, complex + processing of large volumes of information that is most efficiently + processed without user interaction. These operations typically include + time based events (e.g. month-end calculations, notices or + correspondence), periodic application of complex business rules processed + repetitively across very large data sets (e.g. insurance benefit + determination or rate adjustments), or the integration of information that + is received from internal and external systems that typically requires + formatting, validation and processing in a transactional manner into the + system of record. Batch processing is used to process billions of + transactions every day for enterprises. + + Spring Batch is a lightweight, comprehensive batch framework + designed to enable the development of robust batch applications vital for + the daily operations of enterprise systems. Spring Batch builds upon the + productivity, POJO-based development approach, and general ease of use + capabilities people have come to know from the Spring Framework, while + making it easy for developers to access and leverage more advance + enterprise services when necessary. + + Spring Batch provides reusable functions that are essential in + processing large volumes of records, including logging/tracing, + transaction management, job processing statistics, job restart, skip, and + resource management. It also provides more advance technical services and + features that will enable extremely high-volume and high performance batch + jobs though optimization and partitioning techniques. Simple as well as + complex, high-volume batch jobs can leverage the framework in a highly + scalable manner to process significant volumes of information. + + Spring Batch is part of the Spring + Portfolio. + +
+ Spring Batch Architecture + + Spring Batch is designed with extensibility and a diverse group of + end users in mind. The figure below shows a sketch of the layered + architecture that supports the extensibility and ease of use for + end-user developers. + + + + + + + + + Figure 1.1: Batch Execution + Environments + +
+ +
+ Supporting Batch Execution Environments + + Spring Batch Architecture showing potential execution environment + implementations support different platforms and end-user goals from the + same blocks of business logic in the Application Layer. The initial + release provides an Infrastructure layer in the form of low level tools. + There is also a simple batch execution environment with sample jobs, + using the infrastructure in its implementation. The batch execution + environment provides robust features for traceability and management of + the batch lifecycle. A key goal is that the management of the batch + process (locating a job and its input, starting, scheduling, restarting, + and finally processing to created results) should be as easy as possible + for developers. + + The Infrastructure provides the ability to batch operations + together, and to retry an piece of work if there is an exception. Both + requirements have a transactional flavour, and similar concepts are + relevant (propagation, synchronisation). They also both lend themselves + to the template programming model common in Spring, c.f. + TransactionTemplate, JdbcTemplate, + JmsTemplate. + + The Simple Batch Execution environment is the first execution + environment available. It provides a robust set of integrated features + including logging/tracing, transaction management, job processing + statistics, job restart, skip, and resource management to enable the + management of the full lifecycle of traditional batch processing. A + number of sample jobs are packaged with this execution environment and + are described in detail to more clearly articulate usage and + capabilities of the execution environment. + + The runtime dependencies of infrastructure, core and execution are + shown in the figure below. + + + + + + + + + Figure 1.2: Runtime Dependencies + +
+ +
+ Roadmap + + Once the framework is released it can be used immediately to + simplify batch optimisations and automatic retries. The framework is + oriented around application developers not needing to know any details + of the framework - there are a few application developer interfaces that + can be used for convenient construction of data processing pipelines, + but apart from that we support as close to a POJO programming model as + is practical. This is similar to the approach taken in Spring Core in + the area of DAO implementation. + + A Partitioned Batch Execution Environment is also being developed + that will provide an alternate scaling solution. This execution + environment will provide more advance technical services and features to + enable extremely high-volume and high performance batch jobs though + proven optimization and partitioning techniques. Proven scaling + techniques will be provided as partitioned strategies allowing users to + spread the load across a pool of clustered J2EE application servers. + There are also discussions to leverage grid technologies as an alternate + scaling solution. + + Matt Welsh's work shows that SEDA has + enormous benefits over more rigid processing architectures, and + messaging environments provde a lot of resilience out of the box. So we + also want to provide a more SEDA flavoured execution environment, as + well as supporting the more traditional ETL style approach. There might + be a tie in with Mule and/or other ESB tools here, giving the benefit of + a very scalable architecture, where the choice of transport and + distribution strategy can be made as late as possible. The same + application code could be used in principle for a standalone tool + processing a small amount of data, and a massive enterprise-scale + bulk-processing engine. +
+ +
+ Background + + While open source software projects and associated communities + have focused greater attention on web-based and SOA messaging-based + architecture frameworks, there has been a notable lack of focus on + reusable architecture frameworks to accommodate Java-based batch + processing needs, despite continued needs to handle such processing + within enterprise IT environments. The lack of a standard, reusable + batch architecture has resulted in the proliferation of many one-off, + in-house solutions developed within client enterprise IT + functions. + + Interface21 and Accenture are collaborating to change this. + Accenture's hands-on industry and technical experience in implementing + batch architectures, Interface21's depth of technical experience, and + Spring's proven programming model together mark a natural and powerful + partnership to create high-quality, market relevant software aimed at + filling an important gap in enterprise Java. Both companies are also + currently working with a number of clients solving similar problems + developing Spring-based batch architecture solutions. This has provided + some useful additional detail and real-life constraints helping to + ensure the solution can be applied to the real-world problems posed by + clients. For these reasons and many more, Interface21 and Accenture have + teamed to collaborate on the development of Spring Batch. + + Accenture is contributing previously proprietary batch processing + architecture frameworks -- based upon decades worth of experience in + building batch architectures with the last several generations of + platforms (i.e., COBOL/Mainframe, C++/Unix, and now Java/anywhere) -- to + the Spring Batch project along with committer resources to drive + support, enhancements, and the future roadmap. + + The collaborative effort between Accenture and Interface21 aims to + promote the standardization of software processing approaches, + frameworks, and tools that can be consistently leveraged by enterprise + users when creating batch applications. Companies and government + agencies desiring to deliver standard, proven solutions to their + enterprise IT environments will benefit from Spring Batch. +
+
+ +
+ Usage Scenarios + + Spring Batch provides a technical framework and programming model to + support long-running processes that perform a given set of tasks + repetitively. A typical batch program generally reads a large number of + records from a database, file, or queue, processes the data in some + fashion, and then writes back data in a modified form. Spring Batch + automates this basic batch iteration, providing the capability to process + similar transactions as a set, typically in an offline environment without + any user interaction. Batch jobs are part of most IT projects and Spring + Batch is the only open source framework that provides a robust, + enterprise-scale solution. Batch processing is an application style for + many enterprise data processing pipelines (e.g. payment and settlement + systems), and the lack of a standard architecture has led many projects to + create their own custom architecture at significant development and + maintenance costs. + + Business Scenarios + + Commit batch process periodically + + + + Concurrent batch processing: parallel processing of a + job + + + + Staged, enterprise message-driven processing + + + + Massively parallel batch processing + + + + Manual or scheduled restart after failure + + + + Sequential processing of dependent steps (with extensions to + workflow-driven batches) + + + + Partial processing: skip records (e.g. on rollback) + + + + Whole-batch transaction: for cases with a simple enough data + model or a small batch size + + + + Technical Objectives + + Batch developers use the Spring programming model: concentrate + on business logic; let the framework take care of + infrastructure. + + + + Clear separation of concerns between the infrastructure, the + batch execution environment, and the batch application. + + + + Provide common, core execution services as interfaces that all + projects can implement. + + + + Provide simple and default implementations of the core + execution interfaces that can be used ‘out of the box’. + + + + Easy to configure, customize, and extend services, by + leveraging the spring framework in all layers. + + + + All existing execution environment services should be easy to + replace or extend, without any impact to the infrastructure + layer. + + + + Provide a simple deployment model, with the architecture JARs + completely separate from the application, built using Maven. + + +
+ +
+ How To Get Started + + There are a number of sample applications that can be used to get + started with Spring Batch. They can be found in the samples project. They + are executed either from the command line or as unit tests. See Chapter 6: Practical Examples for Spring Batch + as a starting point. +
+
\ No newline at end of file diff --git a/docs/src/site/docbook/reference/spring-tasklet.xml b/docs/src/site/docbook/reference/spring-tasklet.xml new file mode 100644 index 000000000..19e0e896f --- /dev/null +++ b/docs/src/site/docbook/reference/spring-tasklet.xml @@ -0,0 +1,30 @@ + + + Tasklet and the Repeat Template + +
+ What is a Tasklet? + Document what a Tasklet is. Include the diagram of showing its place in the batch domain world. +
+ +
+ Item Providers + Document Item Providers. Discuss some default item providers +
+ +
+ Item Processor + Talk about default Item Processor +
+ +
+ Item Provider Process Tasklet + Don't worry about how all the pieces work but focus on Talk about default Item Processor +
+ +
+ The role of the Repeat Template + The two repeat templates used by ItemProcessor. +
+
diff --git a/docs/src/site/resources/reference/images/BatchExecutionEnvironments.bmp b/docs/src/site/resources/reference/images/BatchExecutionEnvironments.bmp new file mode 100644 index 000000000..87881d477 Binary files /dev/null and b/docs/src/site/resources/reference/images/BatchExecutionEnvironments.bmp differ diff --git a/docs/src/site/resources/reference/images/ExecutionEnvironment.png b/docs/src/site/resources/reference/images/ExecutionEnvironment.png new file mode 100644 index 000000000..e574236b3 Binary files /dev/null and b/docs/src/site/resources/reference/images/ExecutionEnvironment.png differ diff --git a/docs/src/site/resources/reference/images/PipeAndFilter.jpg b/docs/src/site/resources/reference/images/PipeAndFilter.jpg new file mode 100644 index 000000000..82d195bfb Binary files /dev/null and b/docs/src/site/resources/reference/images/PipeAndFilter.jpg differ diff --git a/docs/src/site/resources/reference/images/PipeAndFilter.png b/docs/src/site/resources/reference/images/PipeAndFilter.png new file mode 100644 index 000000000..07ba32c73 Binary files /dev/null and b/docs/src/site/resources/reference/images/PipeAndFilter.png differ diff --git a/docs/src/site/resources/reference/images/RepeatTemplate.png b/docs/src/site/resources/reference/images/RepeatTemplate.png new file mode 100644 index 000000000..0dbb906e6 Binary files /dev/null and b/docs/src/site/resources/reference/images/RepeatTemplate.png differ diff --git a/docs/src/site/resources/reference/images/RuntimeDependencies.png b/docs/src/site/resources/reference/images/RuntimeDependencies.png new file mode 100644 index 000000000..9df1e733f Binary files /dev/null and b/docs/src/site/resources/reference/images/RuntimeDependencies.png differ diff --git a/docs/src/site/resources/reference/images/execution-environment-config.jpg b/docs/src/site/resources/reference/images/execution-environment-config.jpg new file mode 100644 index 000000000..51848deb3 Binary files /dev/null and b/docs/src/site/resources/reference/images/execution-environment-config.jpg differ diff --git a/docs/src/site/resources/reference/images/flatfile-input-source-diagram.jpg b/docs/src/site/resources/reference/images/flatfile-input-source-diagram.jpg new file mode 100644 index 000000000..df0e08bfc Binary files /dev/null and b/docs/src/site/resources/reference/images/flatfile-input-source-diagram.jpg differ diff --git a/docs/src/site/resources/reference/images/io-design.jpg b/docs/src/site/resources/reference/images/io-design.jpg new file mode 100644 index 000000000..7df5d4015 Binary files /dev/null and b/docs/src/site/resources/reference/images/io-design.jpg differ diff --git a/docs/src/site/resources/reference/images/jmx-job.jpg b/docs/src/site/resources/reference/images/jmx-job.jpg new file mode 100644 index 000000000..7dc09824a Binary files /dev/null and b/docs/src/site/resources/reference/images/jmx-job.jpg differ diff --git a/docs/src/site/resources/reference/images/jmx.jpg b/docs/src/site/resources/reference/images/jmx.jpg new file mode 100644 index 000000000..0e0a3e8c4 Binary files /dev/null and b/docs/src/site/resources/reference/images/jmx.jpg differ diff --git a/docs/src/site/resources/reference/images/nfljob-config.jpg b/docs/src/site/resources/reference/images/nfljob-config.jpg new file mode 100644 index 000000000..8df0ab4fb Binary files /dev/null and b/docs/src/site/resources/reference/images/nfljob-config.jpg differ diff --git a/docs/src/site/resources/reference/images/nfljob.jpg b/docs/src/site/resources/reference/images/nfljob.jpg new file mode 100644 index 000000000..2db5910f5 Binary files /dev/null and b/docs/src/site/resources/reference/images/nfljob.jpg differ diff --git a/docs/src/site/resources/reference/images/s1-job-configuration.jpg b/docs/src/site/resources/reference/images/s1-job-configuration.jpg new file mode 100644 index 000000000..df639d3dd Binary files /dev/null and b/docs/src/site/resources/reference/images/s1-job-configuration.jpg differ diff --git a/docs/src/site/resources/reference/images/simple-batch-execution-env.jpg b/docs/src/site/resources/reference/images/simple-batch-execution-env.jpg new file mode 100644 index 000000000..87a21fbc0 Binary files /dev/null and b/docs/src/site/resources/reference/images/simple-batch-execution-env.jpg differ diff --git a/docs/src/site/resources/reference/images/simple-tasklet-job-configuration.jpg b/docs/src/site/resources/reference/images/simple-tasklet-job-configuration.jpg new file mode 100644 index 000000000..6c223d8b9 Binary files /dev/null and b/docs/src/site/resources/reference/images/simple-tasklet-job-configuration.jpg differ diff --git a/docs/src/site/resources/reference/images/spring-batch-football-graph.jpg b/docs/src/site/resources/reference/images/spring-batch-football-graph.jpg new file mode 100644 index 000000000..d8898c135 Binary files /dev/null and b/docs/src/site/resources/reference/images/spring-batch-football-graph.jpg differ diff --git a/docs/src/site/site.xml b/docs/src/site/site.xml index 86829b1b6..1f08e99dc 100644 --- a/docs/src/site/site.xml +++ b/docs/src/site/site.xml @@ -22,6 +22,11 @@ + + + + + ${reports}