casacore
Loading...
Searching...
No Matches
Tables.h
Go to the documentation of this file.
1// # Tables.h: The Tables module - Casacore data storage
2// # Copyright (C) 1994-2010
3// # Associated Universities, Inc. Washington DC, USA.
4// #
5// # This library is free software; you can redistribute it and/or modify it
6// # under the terms of the GNU Library General Public License as published by
7// # the Free Software Foundation; either version 2 of the License, or (at your
8// # option) any later version.
9// #
10// # This library is distributed in the hope that it will be useful, but WITHOUT
11// # ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or
12// # FITNESS FOR A PARTICULAR PURPOSE. See the GNU Library General Public
13// # License for more details.
14// #
15// # You should have received a copy of the GNU Library General Public License
16// # along with this library; if not, write to the Free Software Foundation,
17// # Inc., 675 Massachusetts Ave, Cambridge, MA 02139, USA.
18// #
19// # Correspondence concerning AIPS++ should be addressed as follows:
20// # Internet email: casa-feedback@nrao.edu.
21// # Postal address: AIPS++ Project Office
22// # National Radio Astronomy Observatory
23// # 520 Edgemont Road
24// # Charlottesville, VA 22903-2475 USA
25
26#ifndef TABLES_TABLES_H
27#define TABLES_TABLES_H
28
29// # Includes
30// # table description
31#include <casacore/casa/aips.h>
32#include <casacore/tables/Tables/TableDesc.h>
33#include <casacore/tables/Tables/ColumnDesc.h>
34#include <casacore/tables/Tables/ScaColDesc.h>
35#include <casacore/tables/Tables/ArrColDesc.h>
36#include <casacore/tables/Tables/ScaRecordColDesc.h>
37
38// # table access
39#include <casacore/tables/Tables/Table.h>
40#include <casacore/tables/Tables/TableLock.h>
41#include <casacore/tables/Tables/SetupNewTab.h>
42#include <casacore/tables/Tables/ScalarColumn.h>
43#include <casacore/tables/Tables/ArrayColumn.h>
44#include <casacore/tables/Tables/TableRow.h>
45#include <casacore/tables/Tables/TableCopy.h>
46#include <casacore/tables/Tables/TableUtil.h>
47#include <casacore/casa/Arrays/Array.h>
48#include <casacore/casa/Arrays/Slicer.h>
49#include <casacore/casa/Arrays/Slice.h>
50
51// # keywords
52#include <casacore/tables/Tables/TableRecord.h>
53#include <casacore/casa/Containers/RecordField.h>
54
55// # table lookup
56#include <casacore/tables/Tables/ColumnsIndex.h>
57#include <casacore/tables/Tables/ColumnsIndexArray.h>
58
59// # table vectors
60#include <casacore/tables/Tables/TableVector.h>
61#include <casacore/tables/Tables/TabVecMath.h>
62#include <casacore/tables/Tables/TabVecLogic.h>
63
64// # data managers
65#include <casacore/tables/DataMan.h>
66
67// # table expressions (for selection of rows)
68#include <casacore/tables/TaQL.h>
69
70namespace casacore { // # NAMESPACE CASACORE - BEGIN
71
72// <module>
73
74// <summary>
75// CTDS (Casacore Table Data System) is the data storage mechanism for Casacore
76// </summary>
77
78// <use visibility=export>
79
80// <reviewed reviewer="jhorstko" date="1994/08/30" tests="" demos="">
81// </reviewed>
82
83// <prerequisite>
84// <li> <linkto class="Record:description">Record</linkto> class
85// </prerequisite>
86
87// <etymology>
88// "Table" is a formal term from relational database theory:
89// <em> "The organizing principle in a relational database is the TABLE,
90// a rectangular, row/column arrangement of data values."</em>
91// Casacore tables are extensions to traditional tables, but are similar
92// enough that we use the same name. There is also a strong resemblance
93// between the uses of Casacore tables, and FITS binary tables, which
94// provides another reason to use "Tables" to describe the Casacore data
95// storage mechanism.
96// </etymology>
97
98// <synopsis>
99// Tables are the fundamental storage mechanism for Casacore. This document
100// explains <A HREF="#Tables:motivation">why</A> they had to be made,
101// <A HREF="#Tables:properties">what</A> their properties are, and
102// <A HREF="#Tables:open">how</A> to use them. The last subject is
103// discussed and illustrated in a sequence of sections:
104// <UL>
105// <LI> <A HREF="#Tables:open">opening</A> an existing table,
106// <LI> <A HREF="#Tables:read">reading</A> from a table,
107// <LI> <A HREF="#Tables:creation">creating</A> a new table,
108// <LI> <A HREF="#Tables:write">writing</A> into a table,
109// <LI> <A HREF="#Tables:row-access">accessing rows</A> in a table,
110// <LI> <A HREF="#Tables:select and sort">selection and sorting</A>
111// (see also <A HREF="../notes/199.html">Table Query Language</A>),
112// <LI> <A HREF="#Tables:concatenation">concatenating similar tables</A>
113// <LI> <A HREF="#Tables:iterate">iterating</A> through a table,
114// <LI> <A HREF="#Tables:LockSync">locking/synchronization</A>
115// for concurrent access,
116// <LI> <A HREF="#Tables:KeyLookup">indexing</A> a table for faster lookup,
117// <LI> <A HREF="#Tables:vectors">vector operations</A> on a column.
118// <LI> <A HREF="#Tables:performance">performance and robustness</A>
119// considerations with some information on
120// <A HREF="#Tables:iotracing">IO tracing</A>.
121// </UL>
122// A few <A HREF="Tables:applications">applications</A> exist to inspect
123// and manipulate a table.
124//
125// Several UML diagrams describe the class structure of the Tables module.
126// <ul>
127// <li> <a href="TableOverview.drawio.svg.html">Global overview of Table access</a>.
128// <li> <a href="TableDesc.drawio.svg.html">Table and column descriptions</a>.
129// <li> <a href="TableRecord.drawio.svg.html">Table keywords</a>.
130// <li> <a href="Table.drawio.svg.html">Table class structure</a>.
131// <li> <a href="PlainTable.drawio.svg.html">Detailed PlainTable class structure</a>.
132// <li> <a href="DataManager.drawio.svg.html">DataManagers for storage</a>.
133// </ul>
134
135// <ANCHOR NAME="Tables:motivation">
136// <motivation></ANCHOR>
137//
138// The Casacore tables are mainly based upon the ideas of Allen Farris,
139// as laid out in the
140// <A HREF="http://aips2.cv.nrao.edu/aips++/docs/reference/Database.ps.gz">
141// AIPS++ Database document</A>, from where the following paragraph is taken:
142//
143// <p>
144// Traditional relational database tables have two features that
145// decisively limit their applicability to scientific data. First, an item of
146// data in a column of a table must be atomic -- it must have no internal
147// structure. A consequence of this restriction is that relational
148// databases are unable to deal with arrays of data items. Second, an
149// item of data in a column of a table must not have any direct or
150// implied linkages to other items of data or data aggregates. This
151// restriction makes it difficult to model complex relationships between
152// collections of data. While these restrictions may make it easy to
153// define a mathematically complete set of data manipulation operations,
154// they are simply intolerable in a scientific data-handling context.
155// Multi-dimensional arrays are frequently the most natural modes in
156// which to discuss and think about scientific data. In addition,
157// scientific data often requires complex calibration operations that
158// must draw on large bodies of data about equipment and its performance
159// in various states. The restrictions imposed by the relational model
160// make it very difficult to deal with complex problems of this nature.
161// <p>
162//
163// In response to these limitations, and other needs, the Casacore tables were
164// designed.
165// </motivation>
166
167// <ANCHOR NAME="Tables:properties">
168// <h3>Table Properties</h3></ANCHOR>
169//
170// Casacore tables have the following properties:
171// <ul>
172// <li> A table consists of a number of rows and columns.
173// <A HREF="#Tables:keywords">Keyword/value pairs</A> may be defined
174// for the table as a whole and for individual columns. A keyword/value
175// pair for a column could, for instance, define its unit.
176// <li> Each table has a <A HREF="#Tables:Table Description">description</A>
177// which specifies the number and type of columns, and maybe initial
178// keyword sets and default values for the columns.
179// <li> A cell in a column may contain
180// <UL>
181// <LI> a scalar;
182// <LI> a "direct" array -- which must have the same shape in all
183// cells of a column, is usually small, and is stored in the
184// table itself;
185// <LI> an "indirect" array -- which may have different shapes in
186// different cells of the same column, is arbitrarily large,
187// and is stored in a separate file;
188// </UL>
189// <li> A column may be
190// <UL>
191// <LI> "filled" -- containing actual data, or
192// <LI> "virtual" -- containing a recipe telling how the data will
193// be generated dynamically
194// </UL>
195// <li> Only the standard Casacore data types can be used in filled
196// columns, be they scalars or arrays: Bool, uChar, Short, uShort,
197// Int, uInt, Int64, float, double, Complex, DComplex and String.
198// Furthermore scalars containing
199// <linkto class=TableRecord>record</linkto> values are possible
200// <li> A column can have a default value, which will automatically be stored
201// in a cell of the column, when a row is added to the table.
202// <li> <A HREF="#Tables:Data Managers">Data managers</A> handle the
203// reading, writing and generation of data. Each column in a table can
204// be assigned its own data manager, which allows for optimization of
205// the data storage per column. The choice of data manager determines
206// whether a column is filled or virtual.
207// <li> Table data are stored in a canonical format, so they can be read
208// on any machine. To avoid needless swapping of bytes, the data can
209// be stored in big endian (as used on e.g. SUN) or little endian
210// (as used on Intel PC-s) canonical format.
211// By default it uses the format specified in the aipsrc variable
212// <code>table.endianformat</code> which defaults to
213// <code>Table::LocalEndian</code> (the endian format of the
214// machine being used when creating the table).
215// <li> The SQL-like
216// <a href="../notes/199.html">Table Query Language</a> (TaQL)
217// can be used to do operations on tables like
218// select, sort, update, insert, delete, and create.
219// </ul>
220//
221// Tables can be in one of four forms:
222// <ul>
223// <li> A plain table is a table stored on disk.
224// It can be shared by multiple processes.
225// <li> A memory table is a table held in memory.
226// It is a process specific table, thus not sharable.
227// The <linkto class=Table>Table::copy</linkto> function can be used
228// to turn a memory table into a plain table.
229// <li> A reference table is a table referencing a plain or memory table.
230// It is the result of a selection or sort on another table.
231// A reference table references the data in the other table, thus
232// changing data in a reference table means that the data in the
233// original table are changed.
234// The <linkto class=Table>Table::deepCopy</linkto> function can be
235// used to turn a reference table into a plain table.
236// <li> <A HREF="#Tables:concatenation">a concatenated table</A>
237// is a union of tables (of any form) with the same description.
238// They are concatenated in a virtual way, thus no copy is made.
239// </ul>
240// Concurrent access from different processes to the same plain table is
241// fully supported by means of a <A HREF="#Tables:LockSync">
242// locking/synchronization</A> mechanism. Concurrent access over NFS is also
243// supported.
244// <p>
245// A (somewhat primitive) mechanism is available to do a
246// <A HREF="#Tables:KeyLookup">table lookup</A> based on the contents
247// of a key.
248
249// <ANCHOR NAME="Tables:open">
250// <h3>Opening an Existing Table</h3></ANCHOR>
251//
252// To open an existing table you just create a
253// <linkto class="Table:description">Table</linkto> object giving
254// the name of the table, like:
255//
256// <srcblock>
257// Table readonly_table ("tableName");
258// // or
259// Table read_and_write_table ("tableName", Table::Update);
260// </srcblock>
261//
262// The constructor option determines whether the table will be opened as
263// readonly or as read/write. A readonly table file must be opened
264// as readonly, otherwise an exception is thrown. The functions
265// <linkto class="Table">Table::isWritable(...)</linkto>
266// can be used to determine if a table is writable.
267//
268// When the table is opened, the data managers are reinstantiated
269// according to their definition at table creation.
270// <p>
271// <ANCHOR NAME="Tables:openTable">
272// The static function <src>TableUtil::openTable</src> can be used to open a table,
273// in particular a subtable, in a simple way by means of the :: notation like
274// <src>maintable::subtable</src>. The :: notation is much better than specifying
275// an explicit path (such as <src>maintable/subtable</src>, because it also works
276// fine if the main table is a reference table (e.g. the result of a selection).
277
278// <ANCHOR NAME="Tables:read">
279// <h3>Reading from a Table</h3></ANCHOR>
280//
281// You can read data from a table column with the "get" functions
282// in the classes
283// <linkto class="ScalarColumn:description">ScalarColumn&lt;T&gt;</linkto>
284// and
285// <linkto class="ArrayColumn:description">ArrayColumn&lt;T&gt;</linkto>.
286// For scalars of a standard data type (i.e. Bool, uChar, Int, Short,
287// uShort, uInt, float, double, Complex, DComplex and String) you could
288// instead use
289// <linkto class="TableColumn">TableColumn::getScalar(...)</linkto> or
290// <linkto class="TableColumn">TableColumn::asXXX(...)</linkto>.
291// These functions offer an extra: they do automatic data type promotion;
292// so that you can, for example, get a double value from a float column.
293//
294// These "get" functions are used in the same way as the simple "put"
295// functions described in the previous section.
296// <p>
297// <linkto class="ScalarColumn:description">ScalarColumn&lt;T&gt;</linkto>
298// can be constructed for a non-writable column. However, an exception
299// is thrown if the put function is used for it.
300// The same is true for
301// <linkto class="ArrayColumn:description">ArrayColumn&lt;T&gt;</linkto> and
302// <linkto class="TableColumn:description">TableColumn</linkto>.
303// <p>
304// A typical program could look like:
305// <srcblock>
306// #include <casacore/tables/Tables/Table.h>
307// #include <casacore/tables/Tables/ScalarColumn.h>
308// #include <casacore/tables/Tables/ArrayColumn.h>
309// #include <casacore/casa/Arrays/Vector.h>
310// #include <casacore/casa/Arrays/Slicer.h>
311// #include <casacore/casa/Arrays/ArrayMath.h>
312// #include <iostream>
313//
314// main()
315// {
316// // Open the table (readonly).
317// Table tab ("some.name");
318//
319// // Construct the various column objects.
320// // Their data type has to match the data type in the table description.
321// ScalarColumn<Int> acCol (tab, "ac");
322// ArrayColumn<Float> arr2Col (tab, "arr2");
323//
324// // Loop through all rows in the table.
325// uInt nrrow = tab.nrow();
326// for (uInt i=0; i<nrow; i++) {
327// // Read the row for both columns.
328// cout << "Column ac in row i = " << acCol(i) << endl;
329// Array<Float> array = arr2Col.get (i);
330// }
331//
332// // Show the entire column ac,
333// // and show the 10th element of arr2 in each row..
334// cout << ac.getColumn();
335// cout << arr2.getColumn (Slicer(Slice(10)));
336// }
337// </srcblock>
338
339// <ANCHOR NAME="Tables:creation">
340// <h3>Creating a Table</h3></ANCHOR>
341//
342// The creation of a table is a multi-step process:
343// <ol>
344// <li>
345// Create a <A HREF="#Tables:Table Description">table description</A>.
346// <li>
347// Create a <linkto class="SetupNewTable:description">SetupNewTable</linkto>
348// object with the name of the new table.
349// <li>
350// Create the necessary <A HREF="#Tables:Data Managers">data managers</A>.
351// <li>
352// Bind each column to the appropriate data manager.
353// The system will bind unbound columns to data managers which
354// are created internally using the default data manager name
355// defined in the column description.
356// <li>
357// Define the shape of direct columns (if that was not already done in the
358// column description).
359// <li>
360// Create the <linkto class="Table:description">Table</linkto>
361// object from the SetupNewTable object. Here, a final check is performed
362// and the necessary files are created.
363// </ol>
364// The recipe above is meant for the creation a plain table, but the
365// creation of a memory table is exactly the same. The only difference
366// is that in call to construct the Table object the Table::Memory
367// type has to be given. Note that in the SetupNewTable object the columns
368// can be bound to any data manager. <src>MemoryTable</src> will rebind
369// stored columns to the <linkto class=MemoryStMan>MemoryStMan</linkto>
370// storage manager, but virtual columns bindings are not changed.
371//
372// The following example shows how you can create a table. An example
373// specifically illustrating the creation of the
374// <A HREF="#Tables:Table Description">table description</A> is given
375// in that section. Other sections discuss the access to the table.
376//
377// <srcblock>
378// #include <casacore/tables/Tables/TableDesc.h>
379// #include <casacore/tables/Tables/SetupNewTab.h>
380// #include <casacore/tables/Tables/Table.h>
381// #include <casacore/tables/Tables/ScaColDesc.h>
382// #include <casacore/tables/Tables/ScaRecordColDesc.h>
383// #include <casacore/tables/Tables/ArrColDesc.h>
384// #include <casacore/tables/Tables/StandardStMan.h>
385// #include <casacore/tables/Tables/IncrementalStMan.h>
386//
387// main()
388// {
389// // Step1 -- Build the table description.
390// TableDesc td("tTableDesc", "1", TableDesc::Scratch);
391// td.comment() = "A test of class SetupNewTable";
392// td.addColumn (ScalarColumnDesc<Int> ("ab" ,"Comment for column ab"));
393// td.addColumn (ScalarColumnDesc<Int> ("ac"));
394// td.addColumn (ScalarColumnDesc<uInt> ("ad","comment for ad"));
395// td.addColumn (ScalarColumnDesc<Float> ("ae"));
396// td.addColumn (ScalarRecordColumnDesc ("arec"));
397// td.addColumn (ArrayColumnDesc<Float> ("arr1",3,ColumnDesc::Direct));
398// td.addColumn (ArrayColumnDesc<Float> ("arr2",0));
399// td.addColumn (ArrayColumnDesc<Float> ("arr3",0,ColumnDesc::Direct));
400//
401// // Step 2 -- Setup a new table from the description.
402// SetupNewTable newtab("newtab.data", td, Table::New);
403//
404// // Step 3 -- Create storage managers for it.
405// StandardStMan stmanStand_1;
406// StandardStMan stmanStand_2;
407// IncrementalStMan stmanIncr;
408//
409// // Step 4 -- First, bind all columns to the first storage
410// // manager. Then, bind a few columns to another storage manager
411// // (which will overwrite the previous bindings).
412// newtab.bindAll (stmanStand_1);
413// newtab.bindColumn ("ab", stmanStand_2);
414// newtab.bindColumn ("ae", stmanIncr);
415// newtab.bindColumn ("arr3", stmanIncr);
416//
417// // Step 5 -- Define the shape of the direct columns.
418// // (this could have been done in the column description).
419// newtab.setShapeColumn( "arr1", IPosition(3,2,3,4));
420// newtab.setShapeColumn( "arr3", IPosition(3,3,4,5));
421//
422// // Step 6 -- Finally, create the table consisting of 10 rows.
423// Table tab(newtab, 10);
424//
425// // Now we can fill the table, which is shown in a next section.
426// // The Table destructor will flush the table to the files.
427// }
428// </srcblock>
429// To create a table in memory, only step 6 has to be modified slightly to:
430// <srcblock>
431// Table tab(newtab, Table::Memory, 10);
432// </srcblock>
433//
434// Note that the function <src>TableUtil::createTable</src> can be used to create a table
435// in a simpler way. It can also be used to create a subtable using the :: notation
436// similar to the <A HREF="#Tables:openTable"><src>Tableutil::openTable</src></A>
437// function described above.
438
439// <ANCHOR NAME="Tables:write">
440// <h3>Writing into a Table</h3></ANCHOR>
441//
442// Once a table has been created or has been opened for read/write,
443// you want to write data into it. Before doing that you may have
444// to add one or more rows to the table.
445// <note role=tip> If a table was created with a given number of rows, you
446// do not need to add rows; you may not even be able to do so.
447// </note>
448//
449// When adding new rows to the table, either via the
450// <linkto class="Table">Table(...) constructor</linkto>
451// or via the
452// <linkto class="Table">Table::addRow(...)</linkto>
453// function, you can choose to have those rows initialized with the
454// default values given in the description.
455//
456// To actually write the data into the table you need the classes
457// <linkto class="ScalarColumn:description">ScalarColumn&lt;T&gt;</linkto> and
458// <linkto class="ArrayColumn:description">ArrayColumn&lt;T&gt;</linkto>.
459// For each column you can construct one or
460// more of these objects. Their put(...) functions
461// let you write a value at a time or the entire column in one go.
462// For arrays you can "put" subsections of the arrays.
463//
464// As an alternative for scalars of a standard data type (i.e. Bool,
465// uChar, Int, Short, uShort, uInt, float, double, Complex, DComplex
466// and String) you could use the functions
467// <linkto class="TableColumn">TableColumn::putScalar(...)</linkto>.
468// These functions offer an extra: automatic data type promotion; so that
469// you can, for example, put a float value in a double column.
470//
471// A typical program could look like:
472// <srcblock>
473// #include <casacore/tables/Tables/TableDesc.h>
474// #include <casacore/tables/Tables/SetupNewTab.h>
475// #include <casacore/tables/Tables/Table.h>
476// #include <casacore/tables/Tables/ScaColDesc.h>
477// #include <casacore/tables/Tables/ArrColDesc.h>
478// #include <casacore/tables/Tables/ScalarColumn.h>
479// #include <casacore/tables/Tables/ArrayColumn.h>
480// #include <casacore/casa/Arrays/Vector.h>
481// #include <casacore/casa/Arrays/Slicer.h>
482// #include <casacore/casa/Arrays/ArrayMath.h>
483// #include <iostream>
484//
485// main()
486// {
487// // First build the table description.
488// TableDesc td("tTableDesc", "1", TableDesc::Scratch);
489// td.comment() = "A test of class SetupNewTable";
490// td.addColumn (ScalarColumnDesc<Int> ("ac"));
491// td.addColumn (ArrayColumnDesc<Float> ("arr2",0));
492//
493// // Setup a new table from the description,
494// // and create the (still empty) table.
495// // Note that since we do not explicitly bind columns to
496// // data managers, all columns will be bound to the default
497// // standard storage manager StandardStMan.
498// SetupNewTable newtab("newtab.data", td, Table::New);
499// Table tab(newtab);
500//
501// // Construct the various column objects.
502// // Their data type has to match the data type in the description.
503// ScalarColumn<Int> ac (tab, "ac");
504// ArrayColumn<Float> arr2 (tab, "arr2");
505// Vector<Float> vec2(100);
506//
507// // Write the data into the columns.
508// // In each cell arr2 will be a vector of length 100.
509// // Since its shape is not set explicitly, it is done implicitly.
510// for (uInt i=0; i<10; i++) {
511// tab.addRow(); // First add a row.
512// ac.put (i, i+10); // value is i+10 in row i
513// indgen (vec2, float(i+20)); // vec2 gets i+20, i+21, ..., i+119
514// arr2.put (i, vec2);
515// }
516//
517// // Finally, show the entire column ac,
518// // and show the 10th element of arr2.
519// cout << ac.getColumn();
520// cout << arr2.getColumn (Slicer(Slice(10)));
521//
522// // The Table destructor writes the table.
523// }
524// </srcblock>
525//
526// In this example we added rows in the for loop, but we could also have
527// created 10 rows straightaway by constructing the Table object as:
528// <srcblock>
529// Table tab(newtab, 10);
530// </srcblock>
531// in which case we would not include
532// <srcblock>
533// tab.addRow()
534// </srcblock>
535//
536// The classes
537// <linkto class="TableColumn:description">TableColumn</linkto>,
538// <linkto class="ScalarColumn:description">ScalarColumn&lt;T&gt;</linkto>, and
539// <linkto class="ArrayColumn:description">ArrayColumn&lt;T&gt;</linkto>
540// contain several functions to put values into a single cell or into the
541// whole column. This may look confusing, but is actually quite simple.
542// The functions can be divided in two groups:
543// <ol>
544// <li>
545// Put the given value into the column cell(s).
546// <ul>
547// <li>
548// The simplest put functions,
549// <linkto class="ScalarColumn">ScalarColumn::put(...)</linkto> and
550// <linkto class="ArrayColumn">ArrayColumn::put(...)</linkto>,
551// put a value into the given column cell. For convenience, there is an
552// <linkto class="ArrayColumn">ArrayColumn::putSlice(...)</linkto>
553// to put only a part of the array.
554// <li>
555// <linkto class="ScalarColumn">ScalarColumn::fillColumn(...)</linkto> and
556// <linkto class="ArrayColumn">ArrayColumn::fillColumn(...)</linkto>
557// fill an entire column by putting the given value into all the cells
558// of the column.
559// <li>
560// The simplest putColumn functions,
561// <linkto class="ScalarColumn">ScalarColumn::putColumn(...)</linkto> and
562// <linkto class="ArrayColumn">ArrayColumn::putColumn(...)</linkto>,
563// put an array of values into the column. There is a special
564// <linkto class="ArrayColumn">ArrayColumn::putColumn(...)</linkto>
565// version which puts only a part of the arrays.
566// </ul>
567//
568// <li>
569// Copy values from another column to this column.<BR>
570// These functions have the advantage that the
571// data type of the input and/or output column can be unknown.
572// The generic TableColumn objects can be used for this purpose.
573// The put(Column) function checks the data types and, if possible,
574// converts them. If the conversion is not possible, it throws an
575// exception.
576// <ul>
577// <li>
578// The put functions copy the value in a cell of the input column
579// to a cell in the output column. The row numbers of the cells
580// in the columns can be different.
581// <li>
582// The putColumn functions copy the entire contents of the input column
583// to the output column. The lengths of the columns must be equal.
584// </ul>
585// Each class has its own set of these functions.
586// <ul>
587// <li>
588// <linkto class="TableColumn">TableColumn::put(...)</linkto> and
589// <linkto class="TableColumn">TableColumn::putColumn(...)</linkto> and
590// are the most generic. They can be
591// used if the data types of both input and output column are unknown.
592// Note that these functions are virtual.
593// <li>
594// <linkto class="ScalarColumn">ScalarColumn::put(...)</linkto>,
595// <linkto class="ArrayColumn">ArrayColumn::put(...)</linkto>,
596// <linkto class="ScalarColumn">ScalarColumn::putColumn(...)</linkto>, and
597// <linkto class="ArrayColumn">ArrayColumn::putColumn(...)</linkto>
598// are less generic and therefore potentially more efficient.
599// The most efficient variants are the ones taking a
600// Scalar/ArrayColumn&lt;T&gt;, because they require no data type
601// conversion.
602// </ul>
603// </ol>
604
605// <ANCHOR NAME="Tables:row-access">
606// <h3>Accessing rows in a Table</h3></ANCHOR>
607//
608// Apart from accessing a table column-wise as described in the
609// previous two sections, it is also possible to access a table row-wise.
610// The <linkto class=TableRow>TableRow</linkto> class makes it possible
611// to access multiple fields in a table row as a whole. Note that like the
612// XXColumn classes described above, there is also an ROTableRow class
613// for access to readonly tables.
614// <p>
615// On construction of a TableRow object it has to be specified which
616// fields (i.e. columns) are part of the row. For these fields a
617// fixed structured <linkto class=TableRecord>TableRecord</linkto>
618// object is constructed as part of the TableRow object. The TableRow::get
619// function will fill this record with the table data for the given row.
620// The user has access to the record and can use
621// <linkto class=RecordFieldPtr>RecordFieldPtr</linkto> objects for
622// speedier access to the record.
623// <p>
624// The class could be used as shown in the following example.
625// <srcblock>
626// // Open the table as readonly and define a row object to contain
627// // the given columns.
628// // Note that the function stringToVector is a very convenient
629// // way to construct a Vector<String>.
630// // Show the description of the fields in the row.
631// Table table("Some.table");
632// ROTableRow row (table, stringToVector("col1,col2,col3"));
633// cout << row.record().description();
634// // Since the structure of the record is known, the RecordFieldPtr
635// // objects could be used to allow for easy and fast access to
636// // the record which is refilled for each get.
637// RORecordFieldPtr<String> col1(row.record(), "col1");
638// RORecordFieldPtr<Double> col2(row.record(), "col2");
639// RORecordFieldPtr<Array<Int> > col3(row.record(), "col3");
640// for (uInt i=0; i<table.nrow(); i++) {
641// row.get (i);
642// someString = *col1;
643// somedouble = *col2;
644// someArrayInt = *col3;
645// }
646// </srcblock>
647// The description of TableRow contains some more extensive examples.
648
649// <ANCHOR NAME="Tables:select and sort">
650// <h3>Table Selection and Sorting</h3></ANCHOR>
651//
652// The result of a select and sort of a table is another table,
653// which references the original table. This means that an update
654// of a sorted or selected table results in the update of the original
655// table. The result is, however, a table in itself, so all table
656// functions (including select and sort) can be used with it.
657// Note that a true copy of such a reference table can be made with
658// the <linkto class=Table>Table::deepCopy</linkto> function.
659// <p>
660// Rows or columns can be selected from a table. Columns can be selected
661// by the
662// <linkto class="Table">Table::project(...)</linkto>
663// function, while rows can be selected by the various
664// <linkto class="Table">Table operator()</linkto> functions.
665// Usually a row is selected by giving a select expression with
666// <linkto class="TableExprNode:description">TableExprNode</linkto>
667// objects. These objects represent the various nodes
668// in an expression, e.g. a constant, a column, or a subexpression.
669// The Table function
670// <linkto class="Table">Table::col(...)</linkto>
671// creates a TableExprNode object for a column. The function
672// <linkto class="Table">Table::key(...)</linkto>
673// does the same for a keyword by reading
674// the keyword value and storing it as a constant in an expression node.
675// All column nodes in an expression must belong to the same table,
676// otherwise an exception is thrown.
677// In the following example we select all rows with RA>10:
678// <srcblock>
679// #include <casacore/tables/Tables/ExprNode.h>
680// Table table ("Table.name");
681// Table result = table (table.col("RA") > 10);
682// </srcblock>
683// while in the next one we select rows with RA and DEC in the given
684// intervals:
685// <srcblock>
686// Table result = table (table.col("RA") > 10
687// && table.col("RA") < 14
688// && table.col("DEC") >= -10
689// && table.col("DEC") <= 10);
690// </srcblock>
691// The following operators can be used to form arbitrarily
692// complex expressions:
693// <ul>
694// <li> Relational operators ==, !=, >, >=, < and <=.
695// <li> Logical operators &&, || and !.
696// <li> Arithmetic operators +, -, *, /, %, and unary + and -.
697// <li> Bit operators ^, &, |, and unary ~.
698// <li> Operator() to take a subsection of an array.
699// </ul>
700// Many functions (like sin, max, conj) can be used in an expression.
701// Class <linkto class=TableExprNode>TableExprNode</linkto> shows
702// the available functions.
703// E.g.
704// <srcblock>
705// Table result = table (sin (table.col("RA")) > 0.5);
706// </srcblock>
707// Function <src>in</src> can be used to select from a set of values.
708// A value set can be constructed using class
709// <linkto class=TableExprNodeSet>TableExprNodeSet</linkto>.
710// <srcblock>
711// TableExprNodeSet set;
712// set.add (TableExprNodeSetElem ("abc"));
713// set.add (TableExprNodeSetElem ("defg"));
714// set.add (TableExprNodeSetElem ("h"));
715// Table result = table (table.col("NAME).in (set));
716// </srcblock>
717// select rows with a NAME equal to <src>abc</src>,
718// <src>defg</src>, or <src>h</src>.
719//
720// <p>
721// You can sort a table on one or more columns containing scalars.
722// In this example we simply sort on column RA (default is ascending):
723// <srcblock>
724// Table table ("Table.name");
725// Table result = table.sort ("RA");
726// </srcblock>
727// Multiple
728// <linkto class="Table">Table::sort(...)</linkto>
729// functions exist which allow for more flexible control over the sort order.
730// In the next example we sort first on RA in descending order
731// and then on DEC in ascending order:
732// <srcblock>
733// Table table ("Table.name");
734// Block<String> sortKeys(2);
735// Block<int> sortOrders(2);
736// sortKeys(0) = "RA";
737// sortOrders(0) = Sort::Descending;
738// sortKeys(1) = "DEC";
739// sortOrders(1) = Sort::Ascending;
740// Table result = table.sort (sortKeys, sortOrders);
741// </srcblock>
742//
743// Tables stemming from the same root, can be combined in several
744// ways with the help of the various logical
745// <linkto class="Table">Table operators</linkto> (operator|, etc.).
746
747// <h4>Table Query Language</h4>
748// The selection and sorting mechanism described above can only be used
749// in a hard-coded way in a C++ program.
750// There is, however, another way. Strings containing selection and
751// sorting commands can be used.
752// The syntax of these commands is based on SQL and is described in the
753// <a href="../notes/199.html">Table Query Language</a> (TaQL) note 199.
754// The language supports UDFs (User Defined Functions) in dynamically
755// loadable libraries as explained in the note.
756// <br>A TaQL command can be executed with the static function
757// <src>tableCommand</src> defined in class
758// <linkto class=TableParse>TableParse</linkto>.
759
760// <ANCHOR NAME="Tables:concatenation">
761// <h3>Table Concatenation</h3></ANCHOR>
762// Tables with identical descriptions can be concatenated in a virtual way
763// using the Table concatenation constructor. Such a Table object behaves
764// as any other Table object, thus any operation can be performed on it.
765// An identical description means that the number of columns, the column names,
766// and their data types of the columns must be the same. The columns do not
767// need to be ordered in the same way nor to be stored in the same way.
768// <br>Note that if tables have different column names, it is possible
769// to form a projection (as described in the previous section) first
770// to make them appear identical.
771//
772// Sometimes a MeasurementSet is partitioned, for instance in chunks of
773// one hour. All those chunks can be virtually concatenated this way.
774// Note that all tables in the concatenation will be opened, thus one might
775// run out of file descriptors if there are many chunks.
776//
777// Similar to reference tables, it is possible to make a concatenated Table
778// persistent by using the <src>rename</src> function. It will not copy the
779// data; only the names of the tables used are written.
780//
781// The keywords of a concatenated table are taken from the first table.
782// It is possible to change or add keywords, but that is not persistent,
783// not even if the concatenated table is made persistent.
784// <br>The keywords holding subtables can be handled in a special way.
785// Normally the subtables of the concatenation are the subtables of the first
786// table are used, but is it possible to concatenate subtables as well by
787// giving their names in the constructor.
788// In this way the, say, SYSCAL subtable of a MeasurementSet can be
789// concatenated as well.
790// <srcblock>
791// // Create virtual concatenation of ms0 and ms1.
792// Block<String> names(2);
793// names[0] = "ms0";
794// names[1] = "ms1";
795// // Also concatenate their SYSCAL subtables.
796// Block<String> subNames(1, "SYSCAL");
797// Table concTab (names, subNames);
798// </srcblock>
799
800// <ANCHOR NAME="Tables:iterate">
801// <h3>Table Iterators</h3></ANCHOR>
802//
803// You can iterate through a table in an arbitrary order by getting
804// a subset of the table consisting of the rows in which the iteration
805// columns have the same value.
806// An iterator object is created by constructing a
807// <linkto class="TableIterator:description">TableIterator</linkto>
808// object with the appropriate column names.
809//
810// In the next example we define an iteration on the columns Time and
811// Baseline. Each iteration step returns a table subset in which Time and
812// Baseline have the same value.
813//
814// <srcblock>
815// // Iterate over Time and Baseline (by default in ascending order).
816// // Time is the main iteration order, thus the first column specified.
817// Table t;
818// Table tab ("UV_Table.data");
819// Block<String> iv0(2);
820// iv0[0] = "Time";
821// iv0[1] = "Baseline";
822// //
823// // Create the iterator. This will prepare the first subtable.
824// TableIterator iter(tab, iv0);
825// Int nr = 0;
826// while (!iter.pastEnd()) {
827// // Get the first subtable.
828// // This will contain rows with equal Time and Baseline.
829// t = iter.table();
830// cout << t.nrow() << " ";
831// nr++;
832// // Prepare the next subtable with the next Time,Baseline value.
833// iter.next();
834// }
835// cout << endl << nr << " iteration steps" << endl;
836// </srcblock>
837//
838// You can define more than one iterator on the same table; they operate
839// independently.
840//
841// Note that the result of each iteration step is a table in itself which
842// references the original table, just as in the case of a sort or select.
843// This means that the resulting table can be used again in a sort, select,
844// iteration, etc..
845
846// <ANCHOR NAME="Tables:vectors">
847// <h3>Table Vectors</h3></ANCHOR>
848//
849// A table vector makes it possible to treat a column in a table
850// as a vector. Almost all operators and functions defined for normal
851// vectors, are also defined for table vectors. So it is, for instance,
852// possible to add a constant to a table vector. This has the effect
853// that the underlying column gets changed.
854//
855// You can use the templated class
856// <linkto class="TableVector:description">TableVector</linkto>
857// to make a scalar column appear as a (table) vector.
858// Columns containing arrays or tables are not supported.
859// The data type of the TableVector object must match the
860// data type of the column.
861// A table vector can also hold a normal vector so that (temporary)
862// results of table vector operations can be handled.
863//
864// In the following example we double the data in column COL1 and
865// store the result in a temporary table vector.
866// <srcblock>
867// // Create a table vector for column COL1.
868// // Note that if the table is readonly, putting data in the table vector
869// // results in an exception.
870// Table tab ("Table.data");
871// TableVector<Int> tabvec(tab, "COL1");
872// // Multiply it by a constant. Result is kept in a Vector in memory.
873// TableVector<Int> temp = 2 * tabvec;
874// </srcblock>
875//
876// In the next example we double the data in COL1 and put the result back
877// in the column.
878// <srcblock>
879// // Create a table vector for column COL1.
880// // It has to be a TableVector to be able to change the column.
881// Table tab ("Table.data", Table::Update);
882// TableVector<Int> tabvec(tab, "COL1");
883// // Multiply it by a constant.
884// tabvec *= 2;
885// </srcblock>
886
887// <ANCHOR NAME="Tables:keywords">
888// <h3>Table Keywords</h3></ANCHOR>
889//
890// Any number of keyword/value pairs may be attached to the table as a whole,
891// or to any individual column. They may be freely added, retrieved,
892// re-assigned, or deleted. They are, in essence, a self-resizing list of
893// values (any of the primitive types) indexed by Strings (the keyword).
894//
895// A table keyword/value pair might be
896// <srcblock>
897// Observer = Grote Reber
898// Date = 10 october 1942
899// </srcblock>
900// Column keyword/value pairs might be
901// <srcblock>
902// Units = mJy
903// Reference Pixel = 320
904// </srcblock>
905// The class
906// <linkto class="TableRecord:description">TableRecord</linkto>
907// represents the keywords in a table.
908// It is (indirectly) derived from the standard record classes in the class
909// <linkto class="Record:description">Record</linkto>
910
911// <ANCHOR NAME="Tables:Table Description">
912// <h3>Table Description</h3></ANCHOR>
913//
914// A table contains a description of itself, which defines the layout of the
915// columns and the keyword sets for the table and for the individual columns.
916// It may also define initial keyword sets and default values for the columns.
917// Such a default value is automatically stored in a cell in the table column,
918// whenever a row is added to the table.
919//
920// The creation of the table descriptor is the first step in the creation of
921// a new table. The description is part of the table itself, but may also
922// exist in a separate file. This is useful if you need to create a number
923// of tables with the same structure; in other circumstances it probably
924// should be avoided.
925//
926// The public classes to set up a table description are:
927// <ul>
928// <li> <linkto class="TableDesc:description">TableDesc</linkto>
929// -- holds the table description.
930// <li> <linkto class="ColumnDesc:description">ColumnDesc</linkto>
931// -- holds a generic column description.
932// <li> <linkto class="ScalarColumnDesc:description">ScalarColumnDesc&lt;T&gt;
933// </linkto>
934// -- defines a column containing a scalar value.
935// <li> <linkto class="ScalarRecordColumnDesc:description">ScalarRecordColumnDesc;
936// </linkto>
937// -- defines a column containing a scalar record value.
938// <li> <linkto class="ArrayColumnDesc:description">ArrayColumnDesc&lt;T&gt;
939// </linkto>
940// -- defines a column containing an (in)direct array.
941// </ul>
942//
943// Here follows a typical example of the construction of a table
944// description. For more specialized things -- like the definition of a
945// default data manager -- we refer to the descriptions of the above
946// mentioned classes.
947//
948// <srcblock>
949// #include <casacore/tables/Tables/TableDesc.h>
950// #include <casacore/tables/Tables/ScaColDesc.h>
951// #include <casacore/tables/Tables/ArrColDesc.h>
952// #include <casacore/tables/Tables/ScaRecordTabDesc.h>
953// #include <casacore/tables/Tables/TableRecord.h>
954// #include <casacore/casa/Arrays/IPosition.h>
955// #include <casacore/casa/Arrays/Vector.h>
956//
957// main()
958// {
959// // Create a new table description
960// // Define a comment for the table description.
961// // Define some keywords.
962// ColumnDesc colDesc1, colDesc2;
963// TableDesc td("tTableDesc", "1", TableDesc::New);
964// td.comment() = "A test of class TableDesc";
965// td.rwKeywordSet().define ("ra" float(3.14));
966// td.rwKeywordSet().define ("equinox", double(1950));
967// td.rwKeywordSet().define ("aa", Int(1));
968//
969// // Define an integer column ab.
970// td.addColumn (ScalarColumnDesc<Int> ("ab", "Comment for column ab"));
971//
972// // Add a scalar integer column ac, define keywords for it
973// // and define a default value 0.
974// // Overwrite the value of keyword unit.
975// ScalarColumnDesc<Int> acColumn("ac");
976// acColumn.rwKeywordSet().define ("scale" Complex(0,0));
977// acColumn.rwKeywordSet().define ("unit", "");
978// acColumn.setDefault (0);
979// td.addColumn (acColumn);
980// td.rwColumnDesc("ac").rwKeywordSet().define ("unit", "DEG");
981//
982// // Add a scalar string column ad and define its comment string.
983// td.addColumn (ScalarColumnDesc<String> ("ad","comment for ad"));
984//
985// // Now define array columns.
986// // This one is indirect and has no dimensionality mentioned yet.
987// td.addColumn (ArrayColumnDesc<Complex> ("Arr1","comment for Arr1"));
988// // This one is indirect and has 3-dim arrays.
989// td.addColumn (ArrayColumnDesc<Int> ("A2r1","comment for Arr1",3));
990// // This one is direct and has 2-dim arrays with axes length 4 and 7.
991// td.addColumn (ArrayColumnDesc<uInt> ("Arr3","comment for Arr1",
992// IPosition(2,4,7),
993// ColumnDesc::Direct));
994//
995// // Add columns containing records.
996// td.addColumn (ScalarRecordColumnDesc ("Rec1"));
997// }
998// </srcblock>
999
1000// <ANCHOR NAME="Tables:Data Managers">
1001// <h3>Data Managers</h3></ANCHOR>
1002//
1003// Data managers take care of the actual access to the data in a column.
1004// There are two kinds of data managers:
1005// <ol>
1006// <li> <A HREF="#Tables:storage managers">Storage managers</A> --
1007// which store the data as such. They can only handle the standard
1008// data types (Bool,...,String) as discussed in the section about the
1009// <A HREF="#Tables:properties">table properties</A>).
1010// <li> <A HREF="#Tables:virtual column engines">Virtual column engines</A>
1011// -- which manipulate the data.
1012// An engine could be a simple thing like scaling the data (as done
1013// in classic AIPS to reduce data storage), but it could also be an
1014// elaborate thing like applying corrections on-the-fly.
1015// <br>A special engine is VirtualTaQLColumn which can be used to define
1016// the contents of a column by means of a TaQL expression. In particular,
1017// it can be used to define a constant value for the entire column.
1018// But it can also be used to calculate the UVW-coordinates on-the-fly.
1019// <br>An engine must be used when storing data objects with a non-standard type.
1020// It has to break down the object into items with standard data types
1021// which can be stored with a storage manager.
1022// </ol>
1023// In general the user of a table does not need to be aware which
1024// data managers are being used underneath. Only when the table is created
1025// data managers have to be bound to the columns. Thereafter it is
1026// completely transparent.
1027//
1028// Data managers needs to be registered, so they can be found when a table is
1029// opened. All data managers mentioned below are part of the system and
1030// pre-registered.
1031// It is, however, also possible to load data managers on demand. If a data
1032// manager is not registered it is tried to load a shared library with the
1033// part of the data manager name (in lowercase) before a dot or left arrow.
1034// The dot makes it possible to have multiple data managers in a shared library,
1035// while the left arrow is meant for templated data manager classes.
1036// <br>E.g. if <src>BitFlagsEngine<uChar></src> was not registered, the shared
1037// library <src>libbitflagsengine.so</src> (or .dylib) will be loaded. If
1038// successful, its function <src>register_bitflagsengine()</src> will be
1039// executed which should register the data manager(s). Thereafter it is known
1040// and will be used. For example in a file Register.h and Register.cc:
1041// <srcblock>
1042// // Declare in .h file as C function, so no name mangling is done.
1043// extern "C" {
1044// void register_bitflagsengine();
1045// }
1046// // Implement in .cc file.
1047// void register_bitflagsengine()
1048// {
1049// BitFlagsEngine<uChar>::registerClass();
1050// BitFlagsEngine<Short>::registerClass();
1051// BitFlagsEngine<Int>::registerClass();
1052// }
1053// </srcblock>
1054// There are several functions that can give information which data managers
1055// are used for which columns and to obtain the characteristics and properties
1056// of them. Class RODataManAccessor and derived classes can be used for it
1057// as well as the functions <src>dataManagerInfo</src> and
1058// <src>showStructure</src> in class Table.
1059
1060// <ANCHOR NAME="Tables:storage managers">
1061// <h3>Storage Managers</h3></ANCHOR>
1062//
1063// Storage managers are used to store the data contained in the column cells.
1064// At table construction time the binding of columns to storage managers is done.
1065// <br>Each storage manager uses one or more files (usually called table.fi_xxx
1066// where i is a sequence number and _xxx is some kind of extension).
1067// Typically several file are used to store the data of the columns of a table.
1068// <br>In order to reduce the number of files (and to support large block sizes),
1069// it is possible to have a single container file (a MultiFile) containing all
1070// data files used by the storage managers. Such a file is called table.mf.
1071// Note that the program <em>lsmf</em> can be used to see which
1072// files are contained in a MultiFile. The program <em>tomf</em> can
1073// convert the files in a MultiFile to regular files.
1074// <br>At table creation time it is decided if a MultiFile will be used. It
1075// can be done by means of the StorageOption object given to the SetupNewTable
1076// constructor and/or by the aipsrc variables:
1077// <ul>
1078// <li> <src>table.storage.option</src> which can have the value
1079// 'multifile', 'sepfile' (meaning separate files), or 'default'.
1080// Currently the default is to use separate files.
1081// <li> <src>table.storage.blocksize</src> defines the block size to be
1082// used by a MultiFile. If 0 is given, the file system's block size
1083// will be used.
1084// </ul>
1085// About all standard storage managers support the MultiFile.
1086// The exception is StManAipsIO, because it is hardly ever used.
1087//
1088// Several storage managers exist, each with its own storage characteristics.
1089// The default and preferred storage manager is <src>StandardStMan</src>.
1090// Other storage managers should only be used if they pay off in
1091// file space (like <src>IncrementalStMan</src> for slowly varying data)
1092// or access speed (like the tiled storage managers for large data arrays).
1093// <br>The storage managers store the data in a big or little endian
1094// canonical format. The format can be specified when the table is created.
1095// By default it uses the endian format as specified in the aipsrc variable
1096// <code>table.endianformat</code> which can have the value local, big,
1097// or little. The default is local.
1098// <ol>
1099// <li>
1100// <linkto class="StandardStMan:description">StandardStMan</linkto>
1101// stores all the values in so-called buckets (equally sized chunks
1102// in the file). It requires little memory.
1103// <br>It replaces the old <src>StManAipsIO</src>.
1104//
1105// <li>
1106// <linkto class="IncrementalStMan:description">IncrementalStMan</linkto>
1107// uses a storage mechanism resembling "incremental backups". A value
1108// is only stored if it is different from the previous row. It is
1109// very well suited for slowly varying data.
1110// <br>The class <linkto class="ROIncrementalStManAccessor:description">
1111// ROIncrementalStManAccessor</linkto> can be used to tune the
1112// behaviour of the <src>IncrementalStMan</src>. It contains functions
1113// to deal with the cache size and to show the behaviour of the cache.
1114//
1115// <li>
1116// The <a href="#Tables:TiledStMan">Tiled Storage Managers</a>
1117// store the data as a tiled hypercube allowing for more or less equally
1118// efficient data access along all main axes. It can be used for
1119// UV-data as well as for image data.
1120//
1121// <li>
1122// <linkto class="StManAipsIO:description">StManAipsIO</linkto>
1123// uses <src>AipsIO</src> to store the data in the columns.
1124// It supports all table functionality, but its I/O is probably not
1125// as efficient as other storage managers. It also requires that
1126// a large part of the table fits in memory.
1127// <br>It should not be used anymore, because it uses a lot of memory
1128// for larger tables and because it is not very robust in case an
1129// application or system crashes.
1130//
1131// <li>
1132// <linkto class="MemoryStMan:description">MemoryStMan</linkto>
1133// holds the data in memory. It means that data 'stored' with this
1134// storage manager are NOT persistent.
1135// <br>This storage manager is primarily meant for tables held in
1136// memory, but it can also be useful for temporary columns in
1137// normal tables. Note, however, that if a table is accessed
1138// concurrently from multiple processes, MemoryStMan data cannot be
1139// synchronized.
1140//
1141// <li>
1142// @ref dyscostman.DyscoStMan is a class that stores data with lossy
1143// compression. It combines non-linear least-squares quantization and
1144// different kinds of normalizaton. With the typical factor of 4
1145// compression, the loss in accuracy from lossy compression is
1146// negligable. It should only be used for real (non-simulated) data
1147// that is in a Measurement Set.
1148// The method is described in this article:
1149// https://arxiv.org/abs/1609.02019.
1150//
1151// <li>
1152// <linkto class="Adios2StMan:description">Adios2StMan</linkto> uses the
1153// <A HREF="https://github.com/ornladios/ADIOS2">ADIOS2 framework</A> to
1154// store and load column data.
1155// <br>ADIOS2 has several configurable storage backend itself, and this
1156// flexibility is also available via Adios2StMan. This includes, among other
1157// things, storing compressed data, or choosing a different on-disk formats.
1158// <br>This storage manager is also special in that it provides parallel
1159// writing capabilities for MPI processes, so that multiple processes can
1160// write into different sections of the same column concurrently.
1161// </ol>
1162//
1163// The storage manager framework makes it possible to support arbitrary files
1164// as tables. This has been used in a case where a file is filled
1165// by the data acquisition system of a telescope. The file is simultaneously
1166// used as a table using a dedicated storage manager. The table
1167// system and storage manager provide a sync function to synchronize
1168// the processes, i.e. to make CTDS aware of changes
1169// in the file size (thus in the table size) by the filling process.
1170//
1171// <note role=tip>
1172// Not all data managers support all the table functionality. So, the choice
1173// of a data manager can greatly influence the type of operations you can do
1174// on the table as a whole.
1175// For example, if a column uses the tiled storage manager,
1176// it is not possible to delete rows from the table, because that storage
1177// manager will not support deletion of rows.
1178// However, it is always possible to delete all columns of a data
1179// manager in one single call.
1180// </note>
1181
1182// <ANCHOR NAME="Tables:TiledStMan">
1183// <h3>Tiled Storage Manager</h3></ANCHOR>
1184// The Tiled Storage Managers allow one to store the data of
1185// one or more columns in a tiled way. Tiling means
1186// that the data are stored without a preferred order to make access
1187// along the different main axes equally efficient. This is done by
1188// storing the data in so-called tiles (i.e. equally shaped subsets of an
1189// array) to increase data locality. The user can define the tile shape
1190// to optimize for the most frequently used access.
1191// <p>
1192// The Tiled Storage Manager has the following properties:
1193// <ul>
1194// <li> There can be more than one Tiled Storage Manager in
1195// a table; each with its own (unique) name.
1196// <li> Each Tiled Storage Manager can store an
1197// N-dimensional so-called hypercolumn.
1198// Elaborate hypercolumns can be defined using
1199// <linkto file="TableDesc.h#defineHypercolumn">
1200// TableDesc::defineHypercolumn</linkto>).
1201// <br>Note that defining a hypercolumn is only necessary if it
1202// contains multiple columns or if the TiledDataStMan is used.
1203// It means that in practice it is hardly ever needed to define a
1204// hypercolumn.
1205// <br>A hypercolumn consists of up to three types of columns:
1206// <dl>
1207// <dt> Data columns
1208// <dd> contain the data to be stored in a tiled way. This will
1209// be done in tiled hypercubes.
1210// There must be at least one data column.
1211// <br> For example: a table contains UV-data with
1212// data columns "Visibility" and "Weight".
1213// <dt> Coordinate columns
1214// <dd> define the world coordinates of the pixels in the data columns.
1215// Coordinate columns are optional, but if given there must
1216// be N coordinate columns for an N-dimensional hypercolumn.
1217// <br>
1218// For example: the data in the example above is 4-dimensional
1219// and has coordinate columns "Time", "Baseline", "Frequency",
1220// and "Polarization".
1221// <dt> Id columns
1222// <dd> are needed if TiledDataStMan is used.
1223// Different rows in the data columns can be stored in different
1224// hypercubes. The values in the id column(s) uniquely identify
1225// the hypercube a row is stored in.
1226// <br>
1227// For example: the line and continuum data in a MeasurementSet
1228// table need to be stored in 2 different hypercubes (because
1229// their shapes are different (see below)). A column containing
1230// the type (line or continuum) has to be used as an id column.
1231// </dl>
1232// <li> If multiple data columns are used, the shape of their data
1233// must be conforming in each individual row.
1234// If data in different rows have different shapes, they must be
1235// stored in different hypercubes, because a hypercube can only hold
1236// data with conforming shapes.
1237// <br>
1238// Thus in the example above, rows with line data will have conforming
1239// shapes and can be stored in one hypercube. The continuum data
1240// will have another shape and can be stored in another hypercube.
1241// <br>
1242// The storage manager keeps track of the mapping of rows to/from
1243// hypercubes.
1244// <li> Each hypercube can be tiled in its own way. It is not required
1245// that an integer number of tiles fits in the hypercube. The last
1246// tiles will be padded as needed.
1247// <li> The last axis of a hypercube can be extensible. This means that
1248// the size of that axis does not need to be defined when the
1249// hypercube is defined in the storage manager. Instead, the hypercube
1250// can be extended when another chunk of data has to be stored.
1251// This can be very useful in, for example, a (quasi-)realtime
1252// environment where the size of the time axis is not known.
1253// <li> If coordinate columns are defined, they describe the coordinates
1254// of the axes of the hypercubes. Each hypercube has its own set of
1255// coordinates.
1256// <li> Data and id columns have to be stored with the Tiled
1257// Storage Manager. However, coordinate columns do not need to be
1258// stored with the Tiled Storage Manager.
1259// Especially in the case where the coordinates for a hypercube axis
1260// are varying (i.e. dependent on other axes), another storage manager
1261// has to be used (because the Tiled Storage Manager can only
1262// hold constant coordinates).
1263// </ul>
1264// <p>
1265// The following Tiled Storage Managers are available:
1266// <dl>
1267// <dt> <linkto class=TiledShapeStMan:description>TiledShapeStMan</linkto>
1268// <dd> can be seen as a specialization of <src>TiledDataStMan</src>
1269// by using the array shape as the id value.
1270// Similarly to <src>TiledDataStMan</src> it can maintain multiple
1271// hypercubes and store multiple rows in a hypercube, but it is
1272// easier to use, because the special <src>addHypercube</src> and
1273// <src>extendHypercube</src> functions are not needed.
1274// An hypercube is automatically added when a new array shape is
1275// encountered.
1276// <br>
1277// This storage manager could be used for a table with a column
1278// containing line and continuum data, which will result
1279// in 2 hypercubes.
1280// <dt> <linkto class=TiledCellStMan:description>TiledCellStMan</linkto>
1281// <dd> creates (automatically) a new hypercube for each row.
1282// Thus each row of the hypercolumn is stored in a separate hypercube.
1283// Note that the row number serves as the id value. So an id column
1284// is not needed, although there are multiple hypercubes.
1285// <br>
1286// This storage manager is meant for tables where the data arrays
1287// in the different rows are not accessed together. One can think
1288// of a column containing images. Each row contains an image and
1289// only one image is shown at a time.
1290// <dt> <linkto class=TiledColumnStMan:description>TiledColumnStMan</linkto>
1291// <dd> creates one hypercube for the entire hypercolumn. Thus all cells
1292// in the hypercube have to have the same shape and therefore this
1293// storage manager is only possible if all columns in the hypercolumn
1294// have the attribute FixedShape.
1295// <br>
1296// This storage manager could be used for a table with a column
1297// containing images for the Stokes parameters I, Q, U, and V.
1298// By storing them in one hypercube, it is possible to retrieve
1299// the 4 Stokes values for a subset of the image or for an individual
1300// pixel in a very efficient way.
1301// <dt> <linkto class=TiledDataStMan:description>TiledDataStMan</linkto>
1302// <dd> allows one to control the creation and extension of hypercubes.
1303// This is done by means of the class
1304// <linkto class=TiledDataStManAccessor:description>
1305// TiledDataStManAccessor</linkto>.
1306// It makes it possible to store, say, row 0-9 in hypercube A,
1307// row 10-34 in hypercube B, row 35-54 in hypercube A again, etc..
1308// <br>
1309// The drawback of this storage manager is that its hypercubes are not
1310// automatically extended when adding new rows. The special functions
1311// <src>addHypercube</src> and <src>extendHypercube</src> have to be
1312// used making it somewhat tedious to use.
1313// Therefore this storage manager may become obsolete in the near future.
1314// </dl>
1315// The Tiled Storage Managers have 3 ways to access and cache the data.
1316// Class <linkto class=TSMOption>TSMOption</linkto> can be used to setup an
1317// access choice and use it in a Table constructor.
1318// <ul>
1319// <li> The old way (the only way until January 2010) uses a cache
1320// of its own to keep tiles that might need to be reused. It will always
1321// access entire tiles, even if only a small part is needed.
1322// It is possible to define a maximum cache size. The description of class
1323// <linkto class=ROTiledStManAccessor>ROTiledStManAccessor</linkto>
1324// contains a discussion about the effect of defining a maximum cache
1325// size.
1326// <li> Memory-mapping the data files. In this way the operating system
1327// takes care of the IO and caching. However, the limited address space
1328// may preclude using it for large tables on 32-bit systems.
1329// <li> Use buffered IO and let the kernel's file cache take care of caching.
1330// It will access the data in chunks of the given buffer size, so the
1331// entire tile does not need to be accessed if only a small part is
1332// needed.
1333// </ul>
1334// Apart from reading, all access ways described above can also handle writing
1335// and extending tables. They create fully equal files. Both little and big
1336// endian data can be read or written.
1337
1338// <ANCHOR NAME="Tables:virtual column engines">
1339// <h3>Virtual Column Engines</h3></ANCHOR>
1340//
1341// Virtual column engines are used to implement the virtual (i.e.
1342// calculated-on-the-fly) columns. CTDS provides
1343// an abstract base class (or "interface class")
1344// <linkto class="VirtualColumnEngine:description">VirtualColumnEngine</linkto>
1345// that specifies the protocol for these engines.
1346// The programmer must derive a concrete class to implement
1347// the application-specific virtual column.
1348// <p>
1349// For example: the programmer
1350// needs a column in a table which is the difference between two other
1351// columns. (Perhaps these two other columns are updated periodically
1352// during the execution of a program.) A good way to handle this would
1353// be to have a virtual column in the table, and write a virtual column
1354// engine which knows how to calculate the difference between corresponding
1355// cells of the two other columns. So the result is that accessing a
1356// particular cell of the virtual column invokes the virtual column engine,
1357// which then gets the values from the other two columns, and returns their
1358// difference. This particular example could be done using
1359// <linkto class="VirtualTaQLColumn:description">VirtualTaQLColumn</linkto>.
1360// <p>
1361// Several virtual column engines exist:
1362// <ol>
1363// <li> The class
1364// <linkto class="VirtualTaQLColumn:description">VirtualTaQLColumn</linkto>
1365// makes it possible to define a column as an arbitrary expression of
1366// other columns. It uses the <a href="../notes/199.html">TaQL</a>
1367// CALC command. The virtual column can be a scalar or an array and
1368// can have one of the standard data types supported by CTDS.
1369// <li> The class
1370// <linkto class="BitFlagsEngine:description">BitFlagsEngine</linkto>
1371// maps an integer bit flags column to a Bool column. A read and write mask
1372// can be defined telling which bits to take into account when mapping
1373// to and from Bool (thus when reading or writing the Bool).
1374// <li> The class
1375// <linkto class="CompressFloat:description">CompressFloat</linkto>
1376// compresses a single precision floating point array by scaling the
1377// values to shorts (16-bit integer).
1378// <li> The class
1379// <linkto class="CompressComplex:description">CompressComplex</linkto>
1380// compresses a single precision complex array by scaling the
1381// values to shorts (16-bit integer). In fact, the 2 parts of the complex
1382// number are combined to an 32-bit integer.
1383// <li> The class
1384// <linkto class="CompressComplexSD:description">CompressComplexSD</linkto>
1385// does the same as CompressComplex, but optimizes for the case where the
1386// imaginary part is zero (which is often the case for Single Dish data).
1387// <li> The double templated class
1388// <linkto class="ScaledArrayEngine:description">ScaledArrayEngine</linkto>
1389// scales the data in an array from, for example,
1390// float to short before putting it.
1391// <li> The double templated class
1392// <linkto class="MappedArrayEngine:description">MappedArrayEngine</linkto>
1393// converts the data from one data type to another. Sometimes it might be
1394// needed to store the residual data in an MS in double precision.
1395// Because the imaging task can only handle single precision, this enigne
1396// can be used to map the data from double to single precision.
1397// <li> The double templated class
1398// <linkto class="RetypedArrayEngine:description">RetypedArrayEngine</linkto>
1399// converts the data from one data type to another with the possibility
1400// to reduce the number of dimensions. For example, it can be used to
1401// store an 2-d array of StokesVector objects as a 3-d array of floats
1402// by treating the 4 data elements as an extra array axis. If the
1403// StokesVector class is simple, it can be done very efficiently.
1404// <li> The class
1405// <linkto class="ForwardColumnEngine:description">
1406// ForwardColumnEngine</linkto>
1407// forwards the gets and puts on a row in a column to the same row
1408// in a column with the same name in another table. This provides
1409// a virtual copy of the referenced column.
1410// <li> The class
1411// <linkto class="ForwardColumnIndexedRowEngine:description">
1412// ForwardColumnIndexedRowEngine</linkto>
1413// is similar to <src>ForwardColumnEngine.</src>.
1414// However, instead of forwarding it to the same row it uses a
1415// a column to map its row number to a row number in the referenced
1416// table. In this way multiple rows can share the same data.
1417// This data manager only allows for get operations.
1418// <li> The calibration module has implemented a virtual column engine
1419// to do on-the-fly calibration in a transparent way.
1420// </ol>
1421// To handle arbitrary data types the templated abstract base class
1422// <linkto class="VSCEngine:description">VSCEngine</linkto>
1423// has been written. An example of how to use this class can be
1424// found in the demo program <src>dVSCEngine.cc</src>.
1425
1426// <ANCHOR NAME="Tables:LockSync">
1427// <h3>Table locking and synchronization</h3></ANCHOR>
1428//
1429// Multiple concurrent readers and writers (also via NFS) of a
1430// table are supported by means of a locking/synchronization mechanism.
1431// This mechanism is not very sophisticated in the sense that it is
1432// very coarsely grained. When locking, the entire table gets locked.
1433// A special lock file is used to lock the table. This lock file also
1434// contains some synchronization data.
1435// <p>
1436// Five ways of locking are supported (see class
1437// <linkto class=TableLock>TableLock</linkto>):
1438// <dl>
1439// <dt> TableLock::PermanentLocking(Wait)
1440// <dd> locks the table permanently (from open till close). This means
1441// that one writer OR multiple readers are possible.
1442// <dt> TableLock::AutoLocking
1443// <dd> does the locking automatically. This is the default mode.
1444// This mode makes it possible that a table is shared amongst
1445// processes without the user needing to write any special code.
1446// It also means that a lock is only released when needed.
1447// <dt> TableLock::AutoNoReadLocking
1448// <dd> is similar to AutoLocking. However, no lock is acquired when
1449// reading the table making it possible to read the table while
1450// another process holds a write-lock. It also means that for read
1451// purposes no automatic synchronization is done when the table is
1452// updated in another process.
1453// Explicit synchronization can be done by means of the function
1454// <src>Table::resync</src>.
1455// <dt> TableLock::UserLocking
1456// <dd> requires that the programmer explicitly acquires and releases
1457// a lock on the table. This makes some kind of transaction
1458// processing possible. E.g. set a write lock, add a row,
1459// write all data into the row and release the lock.
1460// The Table functions <src>lock</src> and <src>unlock</src>
1461// have to be used to acquire and release a (read or write) lock.
1462// <dt> TableLock::UserNoReadLocking
1463// <dd> is similar to UserLocking. However, similarly to AutoNoReadLocking
1464// no lock is needed to read the table.
1465// <dt> TableLock::NoLocking
1466// <dd> does not use table locking. It is the responsibility of the
1467// user to ensure that no concurrent access is done on the same
1468// bucket or tile in a storage manager, otherwise a table might
1469// get corrupted.
1470// <br>This mode is always used if Casacore is built with
1471// -DAIPS_TABLE_NOLOCKING.
1472// </dl>
1473// Synchronization of the processes accessing the same table is done
1474// by means of the lock file. When a lock is released, the storage
1475// managers flush their data into the table files. Some synchronization data
1476// is written into the lock file telling the new number of table rows
1477// and telling which storage managers have written data.
1478// This information is read when another process acquires the lock
1479// and is used to determine which storage managers have to refresh
1480// their internal caches.
1481// <br>Note that for the NoReadLocking modes (see above) explicit
1482// synchronization might be needed using <src>Table::resync</src>.
1483// <p>
1484// The function <src>Table::hasDataChanged</src> can be used to check
1485// if a table is (being) changed by another process. In this way
1486// a program can react on it. E.g. the table browser can refresh its
1487// screen when the underlying table is changed.
1488// <p>
1489// In general the default locking option will do.
1490// From the above it should be clear that heavy concurrent access
1491// results in a lot of flushing, thus will have a negative impact on
1492// performance. If uninterrupted access to a table is needed,
1493// the <src>PermanentLocking</src> option should be used.
1494// If transaction-like processing is done (e.g. updating a table
1495// containing an observation catalogue), the <src>UserLocking</src>
1496// option is probably best.
1497// <p>
1498// Creation or deletion of a table is not possible if that table
1499// is still open in another process. The function
1500// <src>Table::isMultiUsed()</src> can be used to check if a table
1501// is open in other processes.
1502// <br>
1503// The function <src>TableUtil::deleteTable</src> should be used to delete
1504// a table. Before deleting the table it ensures that it is writable
1505// and that it is not open in the current or another process.
1506// <p>
1507// The following example wants to read the table uninterrupted, thus it uses
1508// the <src>PermanentLocking</src> option. It also wants to wait
1509// until the lock is actually acquired.
1510// Note that the destructor closes the table and releases the lock.
1511// <srcblock>
1512// // Open the table (readonly).
1513// // Acquire a permanent (read) lock.
1514// // It waits until the lock is acquired.
1515// Table tab ("some.name",
1516// TableLock(TableLock::PermanentLockingWait));
1517// </srcblock>
1518//
1519// The following example uses the automatic locking..
1520// It tells the system to check about every 20 seconds if another
1521// process wants access to the table.
1522// <srcblock>
1523// // Open the table (readonly).
1524// Table tab ("some.name",
1525// TableLock(TableLock::AutoLocking, 20));
1526// </srcblock>
1527//
1528// The following example gets data (say from a GUI) and writes it
1529// as a row into the table. The lock the table as little as possible
1530// the lock is acquired just before writing and released immediately
1531// thereafter.
1532// <srcblock>
1533// // Open the table (writable).
1534// Table tab ("some.name",
1535// TableLock(TableLock::UserLocking),
1536// Table::Update);
1537// while (True) {
1538// get input data
1539// tab.lock(); // Acquire a write lock and wait for it.
1540// tab.addRow();
1541// write data into the row
1542// tab.unlock(); // Release the lock.
1543// }
1544// </srcblock>
1545//
1546// The following example deletes a table if it is not used in
1547// another process.
1548// <srcblock>
1549// Table tab ("some.name");
1550// if (! tab.isMultiUsed()) {
1551// tab.markForDelete();
1552// }
1553// </srcblock>
1554
1555// <ANCHOR NAME="Tables:KeyLookup">
1556// <h3>Table lookup based on a key</h3></ANCHOR>
1557//
1558// Class <linkto class=ColumnsIndex>ColumnsIndex</linkto> offers the
1559// user a means to find the rows matching a given key or key range.
1560// It is a somewhat primitive replacement of a B-tree index and in the
1561// future it may be replaced by a proper B+-tree implementation.
1562// <p>
1563// The <src>ColumnsIndex</src> class makes it possible to build an
1564// in-core index on one or more columns. Looking a key or key range
1565// is done using a binary search on that index. It returns a vector
1566// containing the row numbers of the rows matching the key (range).
1567// <p>
1568// The class is not capable of tracing changes in the underlying column(s).
1569// It detects a change in the number of rows and updates the index
1570// accordingly. However, it has to be told explicitly when a value
1571// in the underlying column(s) changes.
1572// <p>
1573// The following example shows how the class can be used.
1574// <example>
1575// Suppose one has an antenna table with key ANTENNA.
1576// <srcblock>
1577// // Open the table and make an index for column ANTENNA.
1578// Table tab("antenna.tab")
1579// ColumnsIndex colInx(tab, "ANTENNA");
1580// // Make a RecordFieldPtr for the ANTENNA field in the index key record.
1581// // Its data type has to match the data type of the column.
1582// RecordFieldPtr<Int> antFld(colInx.accessKey(), "ANTENNA");
1583// // Now loop in some way and find the row for the antenna
1584// // involved in that loop.
1585// Bool found;
1586// while (...) {
1587// // Fill the key field and get the row number.
1588// // ANTENNA is a unique key, so only one row number matches.
1589// // Otherwise function getRowNumbers had to be used.
1590// *antFld = antenna;
1591// uInt antRownr = colInx.getRowNumber (found);
1592// if (!found) {
1593// cout << "Antenna " << antenna << " is unknown" << endl;
1594// } else {
1595// // antRownr can now be used to get data from that row in
1596// // the antenna table.
1597// }
1598// }
1599// </srcblock>
1600// </example>
1601// <linkto class=ColumnsIndex>ColumnsIndex</linkto> itself contains a more
1602// advanced example. It shows how to use a private compare function
1603// to adjust the lookup if the index does not contain single
1604// key values, but intervals instead. This is useful if a row in
1605// a (sub)table is valid for, say, a time range instead of a single
1606// timestamp.
1607
1608// <ANCHOR NAME="Tables:performance">
1609// <h3>Performance and robustness considerations</h3></ANCHOR>
1610//
1611// CTDS resembles a database system, but it is not as robust.
1612// It lacks the transaction and logging facilities common to data base systems.
1613// It means that in case of a crash data might be lost.
1614// To reduce the risk of data loss to
1615// a minimum, it is advisable to regularly do a <tt>flush</tt>, optionally
1616// with an <tt>fsync</tt> to ensure that all data are really written.
1617// However, that can degrade the performance because it involves extra writes.
1618// So one should find the right balance between robustness and performance.
1619//
1620// To get a good feeling for the performance issues, it is important to
1621// understand some of the internals of CTDS.
1622// <br>The storage managers drive the performance. All storage managers use
1623// buckets (called tiles for the TiledStMan) which contain the data.
1624// All IO is done by bucket. The bucket/tile size is defined when creating
1625// the storage manager objects. Sometimes the default will do, but usually
1626// it is better to set it explicitly.
1627//
1628// It is best to do a flush when a tile is full.
1629// For example: <br>
1630// When creating a MeasurementSet containing N antennae (thus N*(N-1) baselines
1631// or N*(N+1) if auto-correlations are stored as well) it makes sense to
1632// store, say, N/2 rows in a tile and do a flush each time all baselines
1633// are written. In that way tiles are fully filled when doing the flush, so
1634// no extra IO is involved.
1635// <br>Here is some code showing this when creating a MeasurementSet.
1636// The code should speak for itself.
1637// <srcblock>
1638// MS* createMS (const String& msName, int nrchan, int nrant)
1639// {
1640// // Get the MS main default table description.
1641// TableDesc td = MS::requiredTableDesc();
1642// // Add the data column and its unit.
1643// MS::addColumnToDesc(td, MS::DATA, 2);
1644// td.rwColumnDesc(MS::columnName(MS::DATA)).rwKeywordSet().
1645// define("UNIT","Jy");
1646// // Store the DATA and FLAG column in two separate files.
1647// // In this way accessing FLAG only is much cheaper than
1648// // when combining DATA and FLAG.
1649// // All data have the same shape, thus use TiledColumnStMan.
1650// // Also store UVW with TiledColumnStMan.
1651// Vector<String> tsmNames(1);
1652// tsmNames[0] = MS::columnName(MS::DATA);
1653// td.rwColumnDesc(tsmNames[0]).setShape (IPosition(2,itsNrCorr,itsNrFreq));
1654// td.defineHypercolumn("TiledData", 3, tsmNames);
1655// tsmNames[0] = MS::columnName(MS::FLAG);
1656// td.rwColumnDesc(tsmNames[0]).setShape (IPosition(2,itsNrCorr,itsNrFreq));
1657// td.defineHypercolumn("TiledFlag", 3, tsmNames);
1658// tsmNames[0] = MS::columnName(MS::UVW);
1659// td.defineHypercolumn("TiledUVW", 2, tsmNames);
1660// // Setup the new table.
1661// SetupNewTable newTab(msName, td, Table::New);
1662// // Most columns vary slowly and use the IncrStMan.
1663// IncrementalStMan incrStMan("ISMData");
1664// // A few columns use he StandardStMan (set an appropriate bucket size).
1665// StandardStMan stanStMan("SSMData", 32768);
1666// // Store all pol and freq and some rows in a single tile.
1667// // autocorrelations are written, thus in total there are
1668// // nrant*(nrant+1)/2 baselines. Ensure a baseline takes up an
1669// // integer number of tiles.
1670// TiledColumnStMan tiledData("TiledData",
1671// IPosition(3,4,nchan,(nrant+1)/2));
1672// TiledColumnStMan tiledFlag("TiledFlag",
1673// IPosition(3,4,nchan,8*(nrant+1)/2));
1674// TiledColumnStMan tiledUVW("TiledUVW", IPosition(2,3,));
1675// IPosition(2,3,nrant*(nrant+1)/2));
1676// newTab.bindAll (incrStMan);
1677// newTab.bindColumn(MS::columnName(MS::ANTENNA1),stanStMan);
1678// newTab.bindColumn(MS::columnName(MS::ANTENNA2),stanStMan);
1679// newTab.bindColumn(MS::columnName(MS::DATA),tiledData);
1680// newTab.bindColumn(MS::columnName(MS::FLAG),tiledFlag);
1681// newTab.bindColumn(MS::columnName(MS::UVW),tiledUVW);
1682// // Create the MS and its subtables.
1683// // Get access to its columns.
1684// MS* msp = new MeasurementSet(newTab);
1685// // Create all subtables.
1686// // Do this after the creation of optional subtables,
1687// // so the MS will know about those optional sutables.
1688// msp->createDefaultSubtables (Table::New);
1689// return msp;
1690// }
1691// </srcblock>
1692
1693// <h4>Some more performance considerations</h4>
1694// Which storage managers to use and how to use them depends heavily on
1695// the type of data and the access patterns to the data. Here follow some
1696// guidelines:
1697// <ol>
1698// <li> Scalar data can be stored with the StandardStMan (SSM) or
1699// IncrementalStMan (ISM). For slowly varying data (e.g. the TIME column
1700// in a MeasurementSet) it is best to use the ISM. Otherwise the SSM.
1701// Note that very long strings (longer than the bucketsize) can only
1702// be stored with the SSM.
1703// <li> Any number of storage managers can be used. In fact, each column
1704// can have a storage manager of its own resulting in column-wise
1705// stored data which is more and more used in data base systems.
1706// In that way a query or sort on that column is very fast, because
1707// the buckets to read only contain data of that column.
1708// In practice one can decide to combine a few frequently used columns
1709// in a storage manager.
1710// <li> Array data can be stored with any column manager. Small fixed size
1711// arrays can be stored directly with the SSM
1712// (or ISM if not changing much).
1713// However, they can also be stored with a TiledStMan (TSM) as shown
1714// for the UVW column in the example above.
1715// <br> Large arrays should usually be stored with a TSM. However,
1716// if it must be possible to change the shape of an array after it
1717// was stored, the SSM (or ISM) must be used. Note that in that
1718// case a lot of disk space can be wasted, because the SSM and ISM
1719// store the array data at the end of the file if the array got
1720// bigger and do not reuse the old space. The only way to
1721// reclaim it is by making a deep copy of the entire table.
1722// <li> If an array is stored with a TSM, it is important to decide
1723// which TSM to use.
1724// <ol>
1725// <li> The TiledColumnStMan is the most efficient, but only suitable
1726// for arrays having the same shape in the entire column.
1727// <li> The TiledShapeStMan is suitable for columns where the arrays
1728// can have a few shapes.
1729// <li> The TiledCellStMan is suitable for columns where the arrays
1730// can have many different shapes.
1731// </ol>
1732// This is discussed in more detail
1733// <a href="#Tables:TiledStMan">above</a>.
1734// <li> If storing an array with a TSM, it can be very important to
1735// choose the right tile shape. Not only does this define the size
1736// of a tile, but it also defines if access in other directions
1737// than the natural direction can be fast. It is also discussed in
1738// more detail <a href="#Tables:TiledStMan">above</a>.
1739// <li> Columns can be combined in a single TiledStMan. For instance, combining DATA
1740// and FLAG is advantageous if FLAG is always used with DATA. However, if FLAG
1741// is used on its own (e.g. in combination with CORRECTED_DATA), it is better
1742// to separate them, otherwise tiles containing FLAG also contain DATA making the
1743// tiles much bigger, thus more expensive to access.
1744// </ol>
1745//
1746// <ANCHOR NAME="Tables:iotracing">
1747// <h4>IO Tracing</h4></ANCHOR>
1748//
1749// Several forms of tracing can be done to see how the Table I/O performs.
1750// <ul>
1751// <li> On Linux/UNIX systems the <src>strace</src> command can be used to
1752// collect trace information about the physical IO.
1753// <li> The function <src>showCacheStatistics</src> in class
1754// TiledStManAccessor can be used to show the number of actual reads
1755// and writes and the percentage of cache hits.
1756// <li> The software has some options to trace the operations done on
1757// tables. It is possible to specify the columns and/or the operations
1758// to be traced. The following <src>aipsrc</src> variables can be used.
1759// <ul>
1760// <li> <src>table.trace.filename</src> specifies the file to write the
1761// trace output to. If not given or empty, no tracing will be done.
1762// The file name can contain environment variables or a tilde.
1763// <li> <src>table.trace.operation</src> specifies the operations to be
1764// traced. It is a string containing s, r, and/or w where
1765// s means tracing RefTable construction (selection/sort),
1766// r means column reads, and w means column writes.
1767// If empty, only the high level table operations (open, create, close)
1768// will be traced.
1769// <li> <src>table.trace.columntype</src> specifies the types of columns to
1770// be traced. It is a string containing the characters s, a, and/or r.
1771// s means all scalar columns, a all array columns, and r all record
1772// columns. If empty and if <src>table.trace.column</src> is empty,
1773// its default value is a.
1774// <li> <src>table.trace.column</src> specifies names of columns to be
1775// traced. Its value can be one or more glob-like patterns separated
1776// by commas without any whitespace. The default is empty.
1777// For example:
1778// <srcblock>
1779// table.trace.column: *DATA,FLAG,WEIGHT*
1780// </srcblock>
1781// to trace all DATA, the FLAG, and all WEIGHT columns.
1782// </ul>
1783// The trace output is a text file with the following columns
1784// separated by a space.
1785// <ul>
1786// <li> The UTC time the trace line was written (with msec accuracy).
1787// <li> The operation: n(ew), o(pen), c(lose), t(able), r(ead), w(rite),
1788// s(election/sort/iter), p(rojection).
1789// t means an arbitrary table operation as given in the name column.
1790// <li> The table-id (as t=i) given at table creation (new) or open.
1791// <li> The table name, column name, or table operation
1792// (as <src>*oper*</src>).
1793// <src>*reftable*</src> means that the operation is on a RefTable
1794// (thus result of selection, sort, projection, or iteration).
1795// <li> The row or rows to access (* means all rows).
1796// Multiple rows are given as a series of ranges like s:e:i,s:e:i,...
1797// where e and i are only given if applicable (default i is 1).
1798// Note that e is inclusive and defaults to s.
1799// <li> The optional array shape to access (none means scalar).
1800// In case multiple rows are accessed, the last shape value is the
1801// number of rows.
1802// <li> The optional slice of the array in each row as [start][end][stride].
1803// </ul>
1804// Shape, start, end, and stride are given in Fortran-order as
1805// [n1,n2,...].
1806// </ul>
1807
1808// <ANCHOR NAME="Tables:applications">
1809// <h4>Applications to inspect/manipulate a table</h4></ANCHOR>
1810// <ul>
1811// <li><em>showtableinfo</em> shows the structure of a table. It can show:
1812// <ul>
1813// <li> the columns and their format (optionally sorted on name)
1814// <li> the data managers used to store the column data
1815// <li> the table and/or column keywords and their values
1816// <li> recursively the same info of the subtables
1817// </ul>
1818// <li><em>showtablelock</em> if a table is locked or opened and by
1819// which process.
1820// <li><em>lsmf</em> shows the virtual files contained in a MultiFile.
1821// <li><em>tomf</em> copies the given files to a MultiFile.
1822// <li><em>taql</em> can be used to query a table using the
1823// <a href="../notes/199.html">Table Query Language</a> (TaQL).
1824// </ul>
1825//
1826// </synopsis>
1827// </module>
1828
1829} // namespace casacore
1830
1831#endif
For temporary backward namespace compatibility, use casa as alias for casacore.
Definition mainpage.dox:28