<?xml version="1.0" encoding="utf-8" standalone="no"?>
<!DOCTYPE Archive SYSTEM "https://greenstone.org/dtd/Archive/1.0/Archive.dtd">
<Archive>
<Section>
  <Description>
    <Metadata name="gsdldoctype">indexed_doc</Metadata>
    <Metadata name="Language">en</Metadata>
    <Metadata name="Encoding">windows_1252</Metadata>
    <Metadata name="Creator">dg5</Metadata>
    <Metadata name="Title">1997-00 Listing of Working Papers</Metadata>
    <Metadata name="URL">http://C:/Users/anupama/Desktop/GS311_14Aug2023/web/sites/localsite/collect/Word-PDF-Enhanced/tmp/1693990085/word01.html</Metadata>
    <Metadata name="UTF8URL">http://C:/Users/anupama/Desktop/GS311_14Aug2023/web/sites/localsite/collect/Word-PDF-Enhanced/tmp/1693990085/word01.html</Metadata>
    <Metadata name="gsdlsourcefilename">import\word01.doc</Metadata>
    <Metadata name="gsdlsourcefilerenamemethod">url</Metadata>
    <Metadata name="gsdlconvertedfilename">tmp\1693990085\word01.html</Metadata>
    <Metadata name="OrigSource">word01.html</Metadata>
    <Metadata name="Source">word01.doc</Metadata>
    <Metadata name="SourceFile">word01.doc</Metadata>
    <Metadata name="Plugin">WordPlugin</Metadata>
    <Metadata name="FileSize">110080</Metadata>
    <Metadata name="FilenameRoot">word01</Metadata>
    <Metadata name="FileFormat">Word</Metadata>
    <Metadata name="srcicon">_icondoc_</Metadata>
    <Metadata name="srclink_file">doc.doc</Metadata>
    <Metadata name="srclinkFile">doc.doc</Metadata>
    <Metadata name="dc.Title">1997-00 Listing of Working Papers</Metadata>
    <Metadata name="Identifier">HASHeaa2992e081949673150f3</Metadata>
    <Metadata name="lastmodified">1666853534</Metadata>
    <Metadata name="lastmodifieddate">20221027</Metadata>
    <Metadata name="oailastmodified">1693990089</Metadata>
    <Metadata name="oailastmodifieddate">20230906</Metadata>
    <Metadata name="assocfilepath">HASHeaa2.dir</Metadata>
    <Metadata name="gsdlassocfile">doc.doc:application/msword:</Metadata>
  </Description>
  <Content>

&lt;div class=WordSection1&gt;

&lt;p class=MsoTitle&gt;&lt;span lang=EN-US&gt;1997-00 Listing of Working Papers &lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;2000/1&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Using
compression to identify acronyms in text&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Stuart &lt;span
class=SpellE&gt;Yeates&lt;/span&gt;, David Bainbridge, Ian H. &lt;span class=SpellE&gt;Witten&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Text mining is
about looking for patterns in natural language text, and may be defined as the
process of &lt;span class=SpellE&gt;analyzing&lt;/span&gt; text to extract information from
it for particular purposes.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;In previous
work, we claimed that compression is a key technology for text mining, and
backed this up with a study that showed how particular kinds of lexical
tokens—names, dates, locations, &lt;i style='mso-bidi-font-style:normal'&gt;etc.&lt;/i&gt;—can
be identified and located in running text, using compression models to provide
the leverage necessary to distinguish different token types (Witten &lt;i
style='mso-bidi-font-style:normal'&gt;et al.&lt;/i&gt;, 1999)&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;2000/2&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Text &lt;span
class=SpellE&gt;categorization&lt;/span&gt; using compression models&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span class=SpellE&gt;&lt;span
lang=EN-GB&gt;Eibe&lt;/span&gt;&lt;/span&gt;&lt;span lang=EN-GB&gt; Frank, Chang &lt;span class=SpellE&gt;Chui&lt;/span&gt;,
Ian H. &lt;span class=SpellE&gt;Witten&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Text &lt;span
class=SpellE&gt;categorization&lt;/span&gt;, or the assignment of natural language texts
to predefined categories based on their content, is of growing importance as
the volume of information available on the internet continues to overwhelm
us.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The use of predefined categories implies
a “supervised learning” approach to &lt;span class=SpellE&gt;categorization&lt;/span&gt;,
where already-classified articles – which effectively define the categories –
are used as “training data” to build a model that can be used for classifying
new articles that comprise the “test data”.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;This contrasts with “unsupervised” learning, where there is no training
data and clusters of like documents are sought amongst the test articles.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;With supervised learning, meaningful labels
(such as &lt;span class=SpellE&gt;keyphrases&lt;/span&gt;) are attached to the training
documents, and appropriate labels can be assigned automatically to test
documents depending on which category they fall into.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;2000/3&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Reserved for
Sally Jo&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;2000/4&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Interactive
machine learning—letting users build classifiers&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Malcolm Ware, &lt;span
class=SpellE&gt;Eibe&lt;/span&gt; Frank, Geoffrey Holmes, Mark Hall, Ian H. &lt;span
class=SpellE&gt;Witten&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;According to
standard procedure, building a classifier is a fully automated process that
follows data preparation by a domain expert.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;In contrast, &amp;lt;I&amp;gt;interactive&amp;lt;/I&amp;gt;machine learning engages
users in actually generating the classifier themselves.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;This offers a natural way of integrating
background knowledge into the &lt;span class=SpellE&gt;modeling&lt;/span&gt; stage—so long
as interactive tools can be designed that support efficient and effective
communication.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;This paper shows that
appropriate techniques can empower users to create models that compete with
classifiers built by state-of-the-art learning algorithms.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;It demonstrates that users—even users who are
not domain experts—can often construct good classifiers, without any help from
a learning algorithm, using a simple two-dimensional visual interface.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Experiments demonstrate that, not
surprisingly, success hinges on the domain: if a few attributes can support
good predictions, users generate accurate classifiers, whereas domains with
many high-order attribute interactions &lt;span class=SpellE&gt;favor&lt;/span&gt; standard
machine learning techniques.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The future
challenge is to achieve a symbiosis between human user and machine learning
algorithm.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;2000/5&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;KEA: Practical
automatic &lt;span class=SpellE&gt;keyphrase&lt;/span&gt; extraction&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Ian H. &lt;span
class=SpellE&gt;Witten&lt;/span&gt;, Gordon W. &lt;span class=SpellE&gt;Paynter&lt;/span&gt;, &lt;span
class=SpellE&gt;Eibe&lt;/span&gt; Frank, Carl &lt;span class=SpellE&gt;Gutwin&lt;/span&gt;, Craig G.
&lt;span class=SpellE&gt;Nevill&lt;/span&gt;-Manning&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span class=SpellE&gt;&lt;span
lang=EN-GB&gt;Keyphrases&lt;/span&gt;&lt;/span&gt;&lt;span lang=EN-GB&gt; provide semantic metadata
that &lt;span class=SpellE&gt;summarize&lt;/span&gt; and &lt;span class=SpellE&gt;characterize&lt;/span&gt;
documents.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;This paper describes &lt;span
class=SpellE&gt;Kea&lt;/span&gt;, an algorithm for automatically extracting &lt;span
class=SpellE&gt;keyphrases&lt;/span&gt; from text.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;&lt;span class=SpellE&gt;Kea&lt;/span&gt; identifies candidate &lt;span class=SpellE&gt;keyphrases&lt;/span&gt;
using lexical methods, calculates feature values for each candidate, and uses a
machine learning algorithm to predict which candidates are good &lt;span
class=SpellE&gt;keyphrases&lt;/span&gt;.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The
machine learning scheme first builds a prediction model using training
documents with known &lt;span class=SpellE&gt;keyphrases&lt;/span&gt;, and then uses the
model to find &lt;span class=SpellE&gt;keyphrases&lt;/span&gt; in new documents.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;We use a large test corpus to evaluate &lt;span
class=SpellE&gt;Kea's&lt;/span&gt; effectiveness in terms of how many author-assigned &lt;span
class=SpellE&gt;keyphrases&lt;/span&gt; are correctly identified.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;The system is simple, robust, and publicly
available.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;2000/6&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;i style='mso-bidi-font-style:
normal'&gt;&lt;span lang=EN-GB style='font-family:Symbol;mso-ascii-font-family:&quot;Times New Roman&quot;;
mso-hansi-font-family:&quot;Times New Roman&quot;;mso-char-type:symbol;mso-symbol-font-family:
Symbol'&gt;&lt;span style='mso-char-type:symbol;mso-symbol-font-family:Symbol'&gt;m&lt;/span&gt;&lt;/span&gt;&lt;/i&gt;&lt;span
lang=EN-GB&gt;-Charts and Z:&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;&lt;span
class=SpellE&gt;hows&lt;/span&gt;, &lt;span class=SpellE&gt;whys&lt;/span&gt; and &lt;span
class=SpellE&gt;wherefores&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Greg Reeve,
Steve Reeves&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;In this paper we
show, by a series of examples, how the &lt;/span&gt;&lt;i style='mso-bidi-font-style:
normal'&gt;&lt;span lang=EN-GB style='font-family:Symbol;mso-ascii-font-family:&quot;Times New Roman&quot;;
mso-hansi-font-family:&quot;Times New Roman&quot;;mso-char-type:symbol;mso-symbol-font-family:
Symbol'&gt;&lt;span style='mso-char-type:symbol;mso-symbol-font-family:Symbol'&gt;m&lt;/span&gt;&lt;/span&gt;&lt;/i&gt;&lt;span
lang=EN-GB&gt;-chart formalism can be translated into Z.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;We give reasons for why this is an
interesting and sensible thing to do and what it might be used for.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;2000/7&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;One dimensional
non-uniform rational B-splines for animation control&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span class=SpellE&gt;&lt;span
lang=EN-GB&gt;Abdelaziz&lt;/span&gt;&lt;/span&gt;&lt;span lang=EN-GB&gt; &lt;span class=SpellE&gt;Mahoui&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Most 3D
animation packages use graphical representations called motion graphs to
represent the variation in time of the motion parameters.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Many use two-dimensional B-splines as
animation curves because of their power to represent free-form curves.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;In this project, we investigate the
possibility of using One-dimensional Non-Uniform Rational B-&lt;span class=SpellE&gt;Spline&lt;/span&gt;
(NURBS) curves for the interactive construction of animation control
curves.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;One-dimensional NURBS curves
present the potential of solving some problems encountered in motion graphs
when two-dimensional B-splines are used.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;The study focuses on the properties of One-dimensional NURBS
mathematical model.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;It also investigates
the algorithms and shape modification tools devised for two-dimensional curves
and their port to the One-dimensional NURBS model.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;It also looks at the issues related to the
user interface used to interactively modify the shape of the curves.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;2000/8&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Correlation-based
feature selection of discrete and numeric class machine learning&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Mark A. Hall&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Algorithms for
feature selection fall into two broad categories:
&amp;lt;I&amp;gt;wrappers&amp;lt;/I&amp;gt;that use the learning algorithm itself to evaluate
the usefulness of features and &amp;lt;I&amp;gt;filters&amp;lt;/I&amp;gt;that evaluate features
according to heuristics based on general characteristics of the data.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;For application to large databases, filters
have proven to be more practical than wrappers because they are much
faster.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;However, most existing filter
algorithms only work with discrete classification problems.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;This paper describes a fast,
correlation-based filter algorithm that can be applied to continuous and
discrete problems.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The algorithm often
out-performs the well-known &lt;span class=SpellE&gt;ReliefF&lt;/span&gt; attribute
estimator when used as a &lt;span class=SpellE&gt;preprocessing&lt;/span&gt; step for naïve
&lt;span class=SpellE&gt;Bayes&lt;/span&gt;, instance-based learning, decision trees,
locally weighted regression, and model trees.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;It performs more feature selection than &lt;span class=SpellE&gt;ReliefF&lt;/span&gt;
does-reducing the data dimensionality by fifty percent in most cases.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Also, decision and model trees built from the
&lt;span class=SpellE&gt;prepocessed&lt;/span&gt; data are often significantly smaller.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;2000/9&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;A development
environment for predictive modelling in foods&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;G. Holmes, &lt;span
class=SpellE&gt;M.A.&lt;/span&gt; Hall&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;WEKA (Waikato
Environment for Knowledge Analysis) is a comprehensive suite of Java class
libraries that implement many state-of-the-art machine learning/data mining
algorithms.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Non-programmers interact
with the software via a user interface component called the Knowledge Explorer.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Applications
constructed from the WEKA class libraries can be run on any computer with a web
browsing capability, allowing users to apply machine learning techniques to
their own data regardless of computer platform.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;This paper describes the user interface component of the WEKA system in
reference to previous applications in the predictive &lt;span class=SpellE&gt;modeling&lt;/span&gt;
of foods.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;2000/10&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Benchmarking
attribute selection techniques for data mining&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Mark A. Hall,
Geoffrey Holmes&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Data engineering
is generally considered to be a central issue in the development of data mining
applications.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The success of many
learning schemes, in their attempts to construct models of data, hinges on the
reliable identification of a small set of highly predictive attributes.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;The inclusion of irrelevant, redundant and
noisy attributes in the model building process phase can result in poor
predictive performance and increased computation.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Attribute
selection generally involves a combination of search and attribute utility
estimation plus evaluation with respect to specific learning schemes.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;This leads to a large number of possible
permutations and has led to a situation where very few benchmark studies have
been conducted.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;This paper
presents a benchmark comparison of several attribute selection methods.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;All the methods produce an attribute ranking,
a useful devise of isolating the individual merit of an attribute.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Attribute selection is achieved by
cross-validating the rankings with respect to a learning scheme to find the
best attributes.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Results are reported
for a selection of standard data sets and two learning schemes C4.5 and naïve &lt;span
class=SpellE&gt;Bayes&lt;/span&gt;.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;2000/11&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Steve Reeves,
Greg Reeve&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;2000/12&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span class=SpellE&gt;&lt;span
lang=EN-GB&gt;Malika&lt;/span&gt;&lt;/span&gt;&lt;span lang=EN-GB&gt; &lt;span class=SpellE&gt;Mahoui&lt;/span&gt;,
Sally Jo Cunningham&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Transaction logs
are invaluable sources of fine-grained information about users' search &lt;span
class=SpellE&gt;behavior&lt;/span&gt;.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;This paper
compares the searching &lt;span class=SpellE&gt;behavior&lt;/span&gt; of users across two
WWW-accessible digital libraries: the New Zealand Digital Library's Computer
Science Technical Reports collection (CSTR), and the &lt;span class=SpellE&gt;Karlsruhe&lt;/span&gt;
Computer Science Bibliographies (CSBIB) collection.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Since the two collections are designed to
support the same type of users-researchers/students in computer science a
comparative log analysis is likely to uncover common searching preferences for
that user group.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The two collections
differ in their content, however; the CSTR indexes a full text collection,
while the CSBIB is primarily a bibliographic database.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Differences in searching &lt;span class=SpellE&gt;behavior&lt;/span&gt;
between the two systems may indicate the effect of differing search facilities
and content type.&lt;/span&gt;&lt;/p&gt;













&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;99/1&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Lexical
attraction for text compression&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span class=SpellE&gt;&lt;span
lang=EN-GB&gt;Joscha&lt;/span&gt;&lt;/span&gt;&lt;span lang=EN-GB&gt; Bach, Ian H. &lt;span
class=SpellE&gt;Witten&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;New methods of
acquiring structural information in text documents may support better
compression by identifying an appropriate prediction context for each
symbol.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The method of “lexical
attraction” infers syntactic dependency structures from statistical analysis of
large corpora.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We describe the
generation of a lexical attraction model, discuss its application to text
compression, and explore its potential to outperform fixed-context models such
as word-level PPM.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Perhaps the most
exciting aspect of this work is the prospect of using compression as a metric
for structure discovery in text.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;99/2&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Generating rule
sets from model trees&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Geoffrey Holmes,
Mark Hall, &lt;span class=SpellE&gt;Eibe&lt;/span&gt; Frank&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Knowledge discovered
in a database must be represented in a form that is easy to understand.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Small, easy to interpret nuggets of knowledge
from data are one requirement and the ability to induce them from a variety of
data sources is a second.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The literature
is abound with classification algorithms, and in recent years with algorithms
for time sequence analysis, but relatively little has been published on
extracting meaningful information from problems involving continuous classes
(regression).&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Model
trees-decision trees with linear models at the leaf nodes-have recently emerged
as an accurate method for numeric prediction that produces understandable
models.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;However, it is well known that
decision lists-ordered sets of If-Then rules-have the potential to be more compact
and therefore more understandable than their tree counterparts.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;In this paper we
present an algorithm for inducing simple, yet accurate rule sets from model
trees.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The algorithm works by repeatedly
building model trees and selecting the best rule at each iteration.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;It produces rule sets that are, on the whole,
as accurate but smaller than the model tree constructed from the entire &lt;span
class=SpellE&gt;dataset&lt;/span&gt;.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Experimental results for various heuristics which attempt to find a
compromise between rule accuracy and rule coverage are reported.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;We also show empirically that our method
produces more accurate and smaller rule sets than the commercial
state-of-the-art rule learning system Cubist.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;99/3&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;A diagnostic
tool for tree based supervised classification learning algorithms&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Leonard &lt;span
class=SpellE&gt;Trigg&lt;/span&gt;, Geoffrey Holmes&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;The process of
developing applications of machine learning and data mining that employ
supervised classification algorithms includes the important step of knowledge
verification.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Interpretable output is
presented to a user so that they can verify that the knowledge contained in the
output makes sense for the given application.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;As the development of an application is an iterative process it is quite
likely that a user would wish to compare models constructed at various times or
stages.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;One crucial
stage where comparison of models is important is when the accuracy of a model
is being estimated, typically using some form of cross-validation.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;This stage is used to establish an estimate
of how well a model will perform on unseen data.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;This is vital information to present to a
user, but it is also important to show the degree of variation between models
obtained from the entire &lt;span class=SpellE&gt;dataset&lt;/span&gt; and models obtained
during cross-validation.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;In this way it
can be verified that the cross-validation models are at least structurally
aligned with the model garnered from the entire &lt;span class=SpellE&gt;dataset&lt;/span&gt;.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;This paper
presents a diagnostic tool for the comparison of tree-based supervised
classification models.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The method is
adapted from work on approximate tree matching and applied to decision
trees.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The tool is described together
with experimental results on standard &lt;span class=SpellE&gt;datasets&lt;/span&gt;.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;99/4&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Feature
selection for discrete and numeric class machine learning&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Mark A. Hall&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Algorithms for
feature selection fall into two broad categories:
&amp;lt;I&amp;gt;wrappers&amp;lt;/I&amp;gt;use the learning algorithm itself to evaluate the
usefulness of features, while &amp;lt;I&amp;gt;filters&amp;lt;/I&amp;gt;evaluate features
according to heuristics based on general characteristics of the data.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;For application to large databases, filters
have proven to be more practical than wrappers because they are much
faster.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;However, most existing filter
algorithms only work with discrete classification problems.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;This paper
describes a fast, correlation-based filter algorithm that can be applied to
continuous and discrete problems.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Experiments using the new method as a &lt;span class=SpellE&gt;preprocessing&lt;/span&gt;
step for naïve &lt;span class=SpellE&gt;Bayes&lt;/span&gt;, instance-based learning,
decision trees, locally weighted regression, and model trees show it to be an
effective feature selector- it reduces the data in dimensionality by more than
sixty percent in most cases without negatively affecting accuracy.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Also, decision and model trees built from the
pre-processed data are often significantly smaller.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;99/5&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Browsing tree
structures&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Mark &lt;span
class=SpellE&gt;Apperley&lt;/span&gt;, Robert &lt;span class=SpellE&gt;Spence&lt;/span&gt;, Stephen &lt;span
class=SpellE&gt;Hodge&lt;/span&gt;, Michael Chester&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Graphic
representations of tree structures are notoriously difficult to create,
display, and interpret, particularly when the volume of information they
contain, and hence the number of nodes, is large.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;The problem of interactively browsing
information held in tree structures is examined, and the implementation of an
innovative tree browser described.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;This
browser is based on distortion-oriented display techniques and intuitive direct
manipulation interaction.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The tree
layout is automatically generated, but the location and extent of detail shown
is controlled by the user.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;It is
suggested that these techniques could be extended to the browsing of more
general networks.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;99/6&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Facilitating
multiple copy/past operations&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Mark &lt;span
class=SpellE&gt;Apperley&lt;/span&gt;, Jay Baker, Dale Fletcher, Bill Rogers&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Copy and paste,
or cut and paste, using a clipboard or paste buffer has long been the principle
facility provided to users for transferring data between and within GUI
applications.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We argue that this
mechanism can be clumsy in circumstances where several pieces of information
must be moved systematically.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;In two
situations - extraction of data fields from unstructured data found in a
directed search process, and reorganisation of computer program source text -
we present alternative, more natural, user interface facilities to make the
task less onerous, and to provide improved visual feedback during the
operation.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;For the data
extraction task we introduce the Stretchable Selection Tool, a &lt;span
class=SpellE&gt;semi&lt;/span&gt;-transparent overlay augmenting the mouse pointer to
automate paste operations and provide information to prompt the user.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;We describe a prototype implementation that
functions in a collaborative software environment, allowing users to &lt;span
class=SpellE&gt;cooperate&lt;/span&gt; on a multiple copy/paste operation.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;For text reorganisation, we present an
extension to &lt;span class=SpellE&gt;Emacs&lt;/span&gt;, providing similar functionality,
but without the collaborative features.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;99/7&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Automating
iterative tasks with programming by demonstration: a user evaluation&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Gordon W. &lt;span
class=SpellE&gt;Paynter&lt;/span&gt;, Ian H. &lt;span class=SpellE&gt;Witten&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Computer users
often face iterative tasks that cannot be automated using the tools and
aggregation techniques provided by their application program: they end up
performing the iteration by hand, repeating user interface actions over and
over again.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We have implemented an
agent, called Familiar, that can be taught to perform iterative tasks using
programming by demonstration (PBD).&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Unlike other PBD systems, it is domain independent and works with
unmodified, widely-used, applications in a popular operating system.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;In a formal evaluation, we found that users
quickly learned to use the agent to automate iterative tasks.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Generally, the participants preferred to use
multiple selection where possible, but could and did use PBD in situations
involving iteration over many commands, or when other techniques were
unavailable.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;99/8&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;A survey of
software requirements specification practices in the New Zealand software
industry&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Lindsay Groves,
Ray &lt;span class=SpellE&gt;Nickson&lt;/span&gt;, Greg Reeve, Steve Reeves, Mark Utting&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;We report on the
software development techniques used in the New Zealand software industry,
paying particular attention to requirements gathering.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;We surveyed a selection of software companies
with a general questionnaire and then conducted in-depth interviews with four
companies.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Our results show a wide
variety in the kinds of companies undertaking software development, employing a
wide range of software development techniques.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Although our data are not sufficiently detailed to draw statistically
significant conclusions, it appears that larger software development groups
typically have more well-defined software development processes, spend
proportionally more time on requirements gathering, and follow more rigorous
testing regimes.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;99/9&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;The LRU*WWW proxy
cache document replacement algorithm&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Chung-&lt;span
class=SpellE&gt;yi&lt;/span&gt; Chang, Tony &lt;span class=SpellE&gt;McGregor&lt;/span&gt;, Geoffrey
Holmes&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Obtaining good
performance from WWW proxy caches is critically dependent on the document
replacement policy used by the proxy.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;This paper validates the work of other authors by reproducing their
studies of proxy cache document replacement algorithms.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;From this basis a cross-trace study is
mounted.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;This demonstrates that the
performance of most document replacement algorithms is dependent on the type of
workload that they are presented with.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Finally we propose a new algorithm, LRU*, that consistently performs
well across all our traces.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;99/10&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Reduced-error
pruning with significance tests&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span class=SpellE&gt;&lt;span
lang=EN-GB&gt;Eibe&lt;/span&gt;&lt;/span&gt;&lt;span lang=EN-GB&gt; Frank, Ian H. &lt;span
class=SpellE&gt;Witten&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;When building
classification models, it is common practice to prune them to counter spurious
effects of the training data: this often improves performance and reduces model
size.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;&amp;quot;Reduced-error pruning&amp;quot;
is a fast pruning procedure for decision trees that is known to produce small
and accurate trees.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Apart from the data
from which the tree is grown, it uses an independent &amp;quot;pruning&amp;quot; set,
and pruning decisions are based on the model's error rate on this fresh
data.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Recently it has been observed that
reduced-error pruning &lt;span class=SpellE&gt;overfits&lt;/span&gt; the pruning data,
producing unnecessarily large decision trees.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;This paper investigates whether standard statistical significance tests
can be used to counter this phenomenon.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;The problem of &lt;span
class=SpellE&gt;overfitting&lt;/span&gt; to the pruning set highlights the need for
significance testing.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We investigate two
classes of test, &amp;quot;parametric&amp;quot; and &amp;quot;non-parametric.&amp;quot;&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;The standard chi-squared statistic can be
used both in a parametric test and as the basis for a non-parametric
permutation test.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;In both cases it is
necessary to select the significance level at which pruning is applied.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;We show empirically that both versions of the
chi-squared test perform equally well if their significance levels are adjusted
appropriately.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Using a collection of
standard &lt;span class=SpellE&gt;datasets&lt;/span&gt;, we show that significance testing
improves on standard reduced error pruning if the significance level is
tailored to the particular &lt;span class=SpellE&gt;dataset&lt;/span&gt; at hand using
cross-validation, yielding consistently smaller trees that perform at least as
well and sometimes better.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;99/11&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span class=SpellE&gt;&lt;span
lang=EN-GB&gt;Weka&lt;/span&gt;&lt;/span&gt;&lt;span lang=EN-GB&gt;: Practical machine learning
tools and techniques with Java implementations&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Ian H. &lt;span
class=SpellE&gt;Witten&lt;/span&gt;, &lt;span class=SpellE&gt;Eibe&lt;/span&gt; Frank, Len &lt;span
class=SpellE&gt;Trigg&lt;/span&gt;, Mark Hall, Geoffrey Holmes, Sally Jo Cunningham&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;The Waikato
Environment for Knowledge Analysis (Weka) is a comprehensive suite of Java
class libraries that implement many state-of-the-art machine learning and data
mining algorithms.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;&lt;span class=SpellE&gt;Weka&lt;/span&gt;
is freely available on the &lt;span class=SpellE&gt;World-Wide&lt;/span&gt; Web and
accompanies a new text on data mining [1] which documents and fully explains
all the algorithms it contains.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Applications written using the &lt;span class=SpellE&gt;Weka&lt;/span&gt; class
libraries can be run on any computer with a Web browsing capability; this
allows users to apply machine learning techniques to their own data regardless
of computer platform.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;99/12&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Pace Regression&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Yong Wang, Ian
H. &lt;span class=SpellE&gt;Witten&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;This paper
articulates a new method of linear regression, “pace regression”, that
addresses many drawbacks of standard regression reported in the
literature—particularly the subset selection problem.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Pace regression improves on classical ordinary
least squares (OLS) regression by evaluating the effect of each variable and
using a clustering analysis to improve the statistical basis for estimating
their contribution to the overall regression.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;As well as outperforming OLS, it also outperforms—in a remarkably
general sense—other linear &lt;span class=SpellE&gt;modeling&lt;/span&gt; techniques in the
literature, including subset selection procedures, which seek a reduction in
dimensionality that falls out as a natural &lt;span class=SpellE&gt;byproduct&lt;/span&gt;
of pace regression.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The paper defines
six procedures that share the fundamental idea of pace regression, all of which
are theoretically justified in terms of asymptotic performance.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Experiments confirm the performance
improvement over other techniques.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;99/13&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;A
compression-based algorithm for Chinese word segmentation&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;W.J. &lt;span
class=SpellE&gt;Teahan&lt;/span&gt;, &lt;span class=SpellE&gt;Yingying&lt;/span&gt; Wen, &lt;span
class=SpellE&gt;Rodger&lt;/span&gt; &lt;span class=SpellE&gt;McNab&lt;/span&gt;, Ian H. &lt;span
class=SpellE&gt;Witten&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;The Chinese
language is written without using spaces or other word delimiters.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Although a text may be thought of as a
corresponding sequence of words, there is considerable ambiguity in the
placement of boundaries.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Interpreting a
text as a sequence of words is beneficial for some information retrieval and
storage tasks: for example, full-text search, word-based compression, and &lt;span
class=SpellE&gt;keyphrase&lt;/span&gt; extraction.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;We describe a
scheme that infers appropriate positions for word boundaries using an adaptive
language model that is standard in text compression.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;It is trained on a corpus of pre-segmented
text, and when applied to new text, interpolates word boundaries so as to &lt;span
class=SpellE&gt;maximize&lt;/span&gt; the compression obtained.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;This simple and general method performs well
with respect to &lt;span class=SpellE&gt;specialized&lt;/span&gt; schemes for Chinese
language segmentation.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;99/14&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Clustering with
finite data from &lt;span class=SpellE&gt;semi&lt;/span&gt;-parametric mixture
distributions&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Yong Wang, Ian
H. &lt;span class=SpellE&gt;Witten&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Existing
clustering methods for the &lt;span class=SpellE&gt;semi&lt;/span&gt;-parametric mixture
distribution perform well as the volume of data increases.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;However, they all suffer from a serious
drawback in finite-data situations: small outlying groups of data points can be
completely ignored in the clusters that are produced, no matter how far away
they lie from the major clusters.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;This
can result in unbounded loss if the loss function is sensitive to the distance
between clusters.&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;This paper
proposes a new distance-based clustering method that overcomes the problem by
avoiding global constraints.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Experimental results illustrate its superiority to existing methods when
small clusters are present in finite data sets; they also suggest that it is
more accurate and stable than other methods even when there are no small
clusters.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;99/15&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;99/16&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;The &lt;span
class=SpellE&gt;Niupepa&lt;/span&gt; Collection:&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Opening
the blinds on a window to the past&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span class=SpellE&gt;&lt;span
lang=EN-GB&gt;Te&lt;/span&gt;&lt;/span&gt;&lt;span lang=EN-GB&gt; &lt;span class=SpellE&gt;Taka&lt;/span&gt; &lt;span
class=SpellE&gt;Keegan&lt;/span&gt;, Sally Jo Cunningham, Mark &lt;span class=SpellE&gt;Apperley&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;This paper
describes the building of a digital library collection of historic
newspapers.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The newspapers (&lt;span
class=SpellE&gt;&lt;i style='mso-bidi-font-style:normal'&gt;Niupepa&lt;/i&gt;&lt;/span&gt; in &lt;span
class=SpellE&gt;Maori&lt;/span&gt;), which were published in New Zealand during the
period 1842 to 1933, form a unique historical record of the &lt;span class=SpellE&gt;Maori&lt;/span&gt;
language, and of events from an historical perspective.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Images of these newspapers have been
converted to digital form, electronic text extracted from these, and the
collection is now being made available over the Internet as a part of the New
Zealand Digital Library (NZDL) project at the University of Waikato.&lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/1&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Boosting trees
for cost-sensitive classifications&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Kai &lt;span
class=SpellE&gt;Ming&lt;/span&gt; Ting, &lt;span class=SpellE&gt;Zijian&lt;/span&gt; &lt;span
class=SpellE&gt;Zheng&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;This paper
explores two boosting techniques for cost-sensitive tree classification in the
situation where misclassification costs change very often.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Ideally, one would like to have only one
induction, and use the induced model for different misclassification
costs.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Thus, it demands robustness of
the induced model against cost changes.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Combining multiple trees gives robust predictions against this
change.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We demonstrate that ordinary
boosting combined with the minimum expected cost criterion to select the
prediction class is a good solution under this situation.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;We also introduce a variant of the ordinary
boosting procedure which &lt;span class=SpellE&gt;utilizes&lt;/span&gt; the cost
information during training.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We show
that the proposed technique performs better than the ordinary boosting in terms
of misclassification cost.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;However, this
technique requires to induce a set of new trees every time the cost
changes.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Our empirical investigation
also reveals some interesting &lt;span class=SpellE&gt;behavior&lt;/span&gt; of boosting
decision trees for cost-sensitive classification.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/2&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Generating
accurate rule sets without global &lt;span class=SpellE&gt;optimization&lt;/span&gt; &lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span class=SpellE&gt;&lt;span
lang=EN-GB&gt;Eibe&lt;/span&gt;&lt;/span&gt;&lt;span lang=EN-GB&gt; Frank, Ian H. &lt;span
class=SpellE&gt;Witten&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;The two dominant
schemes for rule-learning, C4.5 and RIPPER, both operate in two stages.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;First they induce an initial rule set and
then they refine it using a rather complex &lt;span class=SpellE&gt;optimization&lt;/span&gt;
stage that discards (C4.5) or adjusts (RIPPER) individual rules to make them
work better together.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;In contrast, this
paper shows how good rule sets can be learned one rule at a time, without any
need for global &lt;span class=SpellE&gt;optimization&lt;/span&gt;.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;We present an algorithm for inferring rules
by repeatedly generating partial decision trees, thus combining the two major
paradigms for rule generation-creating rules from decision trees and the
separate-and-conquer rule-learning technique.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;The algorithm is straightforward and elegant: despite this, experiments
on standard &lt;span class=SpellE&gt;datasets&lt;/span&gt; show that it produces rule sets
that are as accurate as and of similar size to those generated by C4.5, and
more accurate than &lt;span class=SpellE&gt;RIPPER's&lt;/span&gt;.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Moreover, it operates efficiently, and
because it avoids &lt;span class=SpellE&gt;postprocessing&lt;/span&gt;, does not suffer the
extremely slow performance on pathological example sets for which the C4.5
method has been &lt;span class=SpellE&gt;criticized&lt;/span&gt;.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/3&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span class=SpellE&gt;&lt;span
lang=EN-GB&gt;VQuery&lt;/span&gt;&lt;/span&gt;&lt;span lang=EN-GB&gt;: a graphical user interface
for Boolean query Specification and dynamic result preview&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Steve Jones&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Textual query
languages based on Boolean logic are common amongst the search facilities of
on-line information repositories.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;However, there is evidence to suggest that the syntactic and semantic
demands of such languages lead to user errors and adversely affect the time
that it takes users to form queries.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Additionally, users are faced with user interfaces to these repositories
which are unresponsive and uninformative, and consequently fail to support
effective query refinement.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We suggest
that graphical query languages, particularly Venn-like diagrams, provide a
natural medium for Boolean query specification which overcomes the problems of
textual query languages.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Also, dynamic
result previews can be seamlessly integrated with graphical query specification
to increase the effectiveness of query refinements.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;We describe &lt;span class=SpellE&gt;VQuery&lt;/span&gt;,
a query interface to the New Zealand Digital Library which exploits querying by
Venn diagrams and integrated query result previews.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/4&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Revising
&amp;lt;I&amp;gt;Z&amp;lt;/I&amp;gt;: semantics and logic&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Martin C. &lt;span
class=SpellE&gt;Henson&lt;/span&gt;, Steve Reeves&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;We introduce a
simple specification logic &amp;lt;I&amp;gt;Z&amp;lt;/I&amp;gt;c comprising a logic and
semantics (in &amp;lt;I&amp;gt;ZF&amp;lt;/I&amp;gt; set theory).&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;We then provide an interpretation for (a
rational reconstruction of) the specification language &amp;lt;I&amp;gt;Z&amp;lt;/I&amp;gt;
within &amp;lt;I&amp;gt;Z&amp;lt;/I&amp;gt;c.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;As a
result we obtain a sound logic for &amp;lt;I&amp;gt;Z&amp;lt;/I&amp;gt;, including the schema
calculus.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;A consequence of our
formalisation is a critique of a number of concepts used in
&amp;lt;I&amp;gt;Z&amp;lt;/I&amp;gt;.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We demonstrate
that the complications and confusions which these concepts introduce can be avoided
without compromising &lt;span class=SpellE&gt;expressibility&lt;/span&gt;.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/5&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;A logic for the
schema calculus&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Martin C. &lt;span
class=SpellE&gt;Henson&lt;/span&gt;, Steve Reeves&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;In this paper we
introduce and investigate a logic for the schema calculus of
&amp;lt;I&amp;gt;Z&amp;lt;/I&amp;gt;.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The schema
calculus is arguably the reason for &amp;lt;I&amp;gt;Z&amp;lt;/I&amp;gt;’s popularity but so
far no true calculus (a sound system of rules for reasoning about schema
expressions) has been given.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Presentations thus far have either failed to provide a calculus (e.g.
the draft standard [3]) or have fallen back on informal descriptions at a
syntactic level (most text books e.g. [7[).&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Once the calculus is established we introduce a derived &lt;span
class=SpellE&gt;equational&lt;/span&gt; logic which enables us to formalise properly the
informal notations of schema expression equality to be found in the literature.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/6&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;New foundations
for &amp;lt;I&amp;gt;Z&amp;lt;/I&amp;gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Martin C. &lt;span
class=SpellE&gt;Henson&lt;/span&gt;, Steve Reeves&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;We provide a
constructive and &lt;span class=SpellE&gt;intensional&lt;/span&gt; interpretation for the
specification language &amp;lt;I&amp;gt;Z&amp;lt;/I&amp;gt; in a theory of operations and kinds
&amp;lt;I&amp;gt;T&amp;lt;/I&amp;gt;.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The motivation is
to facilitate the development of an integrated approach to program
construction.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We illustrate the new
foundations for &amp;lt;I&amp;gt;Z&amp;lt;/I&amp;gt; with examples.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/7&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Predicting apple
bruising relationships using machine learning&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;G. Holmes, S.J.
Cunningham, B.T. &lt;span class=SpellE&gt;Dela&lt;/span&gt; Rue, &lt;span class=SpellE&gt;A.F.&lt;/span&gt;
&lt;span class=SpellE&gt;Bollen&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoBodyText&gt;&lt;span lang=EN-US&gt;Many models have been used to describe
the influence of internal or external factors on apple bruising.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Few of these have addressed the application
of derived relationships to the evaluation of commercial operations.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;From an industry perspective, a model must
enable fruit to be rejected on the basis of a commercially significant bruise
and must also accurately quantify the effects of various combinations of input
features (such as &lt;span class=SpellE&gt;cultivar&lt;/span&gt;, maturity, size, and so
on) on bruise prediction.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Input features
must in turn have characteristics which are measurable commercially; for
example, the measure of force should be impact energy rather than energy
absorbed.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Further, as the commercial
criteria for acceptable damage levels change, the model should be versatile
enough to regenerate new bruise thresholds from existing data.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Machine learning
is a burgeoning technology with a vast range of potential applications
particularly in agriculture where large amounts of data can be readily
collected [1].&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The main advantage of
using a machine learning method in an application is that the models built for
prediction can be viewed and understood by the owner of the data who is in a
position to determine the usefulness of the model, an essential component in a
commercial environment.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/8&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;An evaluation of
passage-level indexing strategies for a technical report archive&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Michael Williams&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Past research
has shown that using evidence from document passages rather than complete
documents is an effective way of improving the precision of full-text database
searches.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;However, passage-level
indexing has yet to be widely adopted for commercial or online databases.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;This paper
reports on experiments designed to test the efficacy of passage-level indexing
with a particular collection of a full-text online database, the New Zealand
Digital Library.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Discourse passages and
word-window passages are used for the indexing process.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Both ranked and Boolean searching are used to
test the resulting indexes.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Overlapping
window passages are shown to offer the best retrieval performance with both
ranked and Boolean queries.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Modifications may be necessary to the term weighting methodology in
order to ensure optimal ranked query performance.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/9&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Managing
multiple collections, multiple languages, and multiple media in a distributed
digital library&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Ian H. &lt;span
class=SpellE&gt;Witten&lt;/span&gt;, &lt;span class=SpellE&gt;Rodger&lt;/span&gt; &lt;span
class=SpellE&gt;McNab&lt;/span&gt;, Steve Jones, Sally Jo Cunningham, David Bainbridge,
Mark &lt;span class=SpellE&gt;Apperley&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Managing the &lt;span
class=SpellE&gt;organizational&lt;/span&gt; and software complexity of a comprehensive
digital library presents a significant challenge.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Different library collections each have their
own distinctive features.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Different
presentation languages have structural implications such as left-to-right
writing order and text-only interfaces for the visually impaired.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Different media involve different file
formats, and-more importantly-radically different search strategies are
required for non-textual media.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;In a
distributed library, new collections can appear asynchronously on servers in
different parts of the world.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;And as
searching interfaces mature from the command-line era exemplified by current
Web search engines into the age of reactive visual interfaces, experimental new
interfaces must be developed, supported, and tested.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;This paper describes our experience, gained
from operating a substantial digital library service over several years, in
solving these problems by designing an appropriate software architecture.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/10&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Experiences with
a weighted decision tree learner&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;John G. &lt;span
class=SpellE&gt;Cleary&lt;/span&gt;, Leonard E. &lt;span class=SpellE&gt;Trigg&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoBodyText&gt;&lt;span lang=EN-US&gt;Machine learning algorithms for inferring
decision trees typically choose a single “best” tree to describe the training
data.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Recent research has shown that
classification performance can be significantly improved by voting predictions
of multiple, independently produced decision trees.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;This paper describes an algorithm, OB1, that
makes a weighted sum over many possible models.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;We describe one instance of OB1, that includes &amp;lt;I&amp;gt;all&amp;lt;/I&amp;gt;
possible decision trees as well as naïve &lt;span class=SpellE&gt;Bayesian&lt;/span&gt;
models.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;OB1 is compared with a number of
other decision tree and instance based learning &lt;span class=SpellE&gt;alogrithms&lt;/span&gt;
on some of the data sets from the UCI repository.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Both an information gain and an accuracy
measure are used for the comparison.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;On
the information gain measure OB1 performs significantly better than all the
other algorithms.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;On the accuracy
measure it is significantly better than all the algorithms except naïve &lt;span
class=SpellE&gt;Bayes&lt;/span&gt; which performs comparably to OB1.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/11&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;An entropy gain
measure of numeric prediction performance&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Leonard &lt;span
class=SpellE&gt;Trigg&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Categorical
classifier performance is typically evaluated with respect to error rate,
expressed as a percentage of test instances that were not correctly
classified.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;When a classifier produces
multiple classifications for a test instance, the prediction is counted as
incorrect (even if the correct class was one of the predictions).&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Although commonly used in the literature,
error rate is a coarse measure of classifier performance, as it is based only
on a single prediction offered for a test instance.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Since many classifiers can produce a class
distribution as a prediction, we should use this to provide a better measure of
how much information the classifier is extracting from the domain.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Numeric
classifiers are a relatively new development in machine learning, and as such
there is no single performance measure that has become standard.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Typically these machine learning schemes
predict a single real number for each test instance, and the error between the
predicted and actual value is used to calculate a myriad of performance
measures such as correlation coefficient, root mean squared error, mean
absolute error, relative absolute error, and root relative squared error.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;With so many performance measures it is
difficult to establish an overall performance evaluation.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;The next section
describes a performance measure for machine learning schemes that attempts to
overcome the problems with current measures.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;In addition, the same evaluation measure is used for categorical and
numeric classifier.&lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/12&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Proceedings of
CBISE ’98 CaiSE*98 Workshop on Component Based Information Systems Engineering&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Edited by John &lt;span
class=SpellE&gt;Grundy&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoBodyText&gt;&lt;span lang=EN-US&gt;Component-based information systems
development is an area of research and practice of increasing importance.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Information Systems developers have &lt;span
class=SpellE&gt;realised&lt;/span&gt; that traditional approaches to IS engineering
produce monolithic, difficult to maintain, difficult to reuse systems.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;In contrast, the use of software components,
which embody data, functionality and well-specified and understood interfaces,
makes interoperable, distributed and highly reusable IS components
feasible.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Component-based approaches to
IS engineering can be used at strategic and &lt;span class=SpellE&gt;organisational&lt;/span&gt;
levels, to model business processes and whole IS architectures, in development
methods which &lt;span class=SpellE&gt;utilise&lt;/span&gt; component-based models during
analysis and design, and in system implementation.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Reusable components can allow end users to
compose and configure their own Information Systems, possibly from a range of
suppliers, and to more tightly couple their &lt;span class=SpellE&gt;organisational&lt;/span&gt;
&lt;span class=SpellE&gt;workflows&lt;/span&gt; with their IS support.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;This workshop
proceedings contains a range of papers addressing one or more of the above
issues relating to the use of component models for IS development.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;All of these papers were refereed by at least
two members of an international workshop committee comprising industry and
academic researchers and users of component technologies.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Strategic uses of components are addressed in
the first three papers, while the following three address uses of components for
systems design and workflow management.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Systems development using components, and the provision of environments
for component management are addressed in the following group of five
papers.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The last three papers in this
proceedings address component management and analysis techniques.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;All of these
papers provide new insights into the many&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;varied uses of component technology for IS engineering.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;I hope you find them as interesting and
useful as I have when collating this proceedings and organising the workshop.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/13&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;An analysis of
usage of a digital library&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Steve Jones,
Sally Jo Cunningham, &lt;span class=SpellE&gt;Rodger&lt;/span&gt; &lt;span class=SpellE&gt;McNab&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;As experimental
digital library &lt;span class=SpellE&gt;testbeds&lt;/span&gt; gain wider acceptance and
develop significant user bases, it becomes important to investigate the ways in
which users interact with the systems in practice.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Transaction logs are one source of usage
information, and the information on user behaviour can be culled from them both
automatically (through calculation of summary statistics) and manually (by
examining query strings for semantic clues on search motivations and searching
strategy).&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We conduct a transaction log
analysis on user activity in the Computer Science Technical Reports Collection
of the New Zealand Digital Library, and report insights gained and identify
resulting search interface design issues.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/14&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Measuring ATM
traffic: final report for New Zealand Telecom&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;John &lt;span
class=SpellE&gt;Cleary&lt;/span&gt;, Ian Graham, Murray Pearson, Tony &lt;span
class=SpellE&gt;McGregor&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;The report
describes the development of a low-cost ATM monitoring system, hosted by a
standard PC.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The monitor can be used
remotely returning information on ATM traffic flows to a central site.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;The monitor is interfaces to a GPS timing
receiver, which provides an absolute time accuracy of better than 1 &lt;span
class=SpellE&gt;usec&lt;/span&gt;.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;By monitoring
the same traffic flow at different points in a network it is possible to
measure cell delay and delay variation in real time, and with existing
traffic.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The monitoring system
characterises cells by a CRC calculated over the cell payload, thus special
measurement cells are not required.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Delays in both local area and wide-area networks have been measured
using this system.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;It is possible to
measure delay in a network that is not end-to-end ATM, as long as some cells
remain identical at the entry and exit points.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Examples are given of traffic and delay measurements in both wide and
local area network systems, including delays measured over the Internet from
Canada to New Zealand.&lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/15&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Despite its
simplicity, the naïve &lt;span class=SpellE&gt;Bayes&lt;/span&gt; learning scheme performs
well on most classification tasks, and is often significantly more accurate
than more sophisticated methods.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Although the probability estimates that it produces can be inaccurate,
it often assigns maximum probability to the correct class.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;This suggests that its good performance might
be restricted to situations where the output is categorical.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;It is therefore interesting to see how it
performs in domains where the predicted value is numeric, because in this case,
predictions are more sensitive to inaccurate probability estimates.&amp;lt;P&amp;gt;&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;This paper shows
how to apply the naïve &lt;span class=SpellE&gt;Bayes&lt;/span&gt; methodology to numeric
prediction (i.e. regression) tasks, and compares it to linear regression,
instance-based learning, and a method that produces “model trees”-decision
trees with linear regression functions at the leaves.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Although we exhibit an artificial &lt;span
class=SpellE&gt;dataset&lt;/span&gt; for which naïve &lt;span class=SpellE&gt;Bayes&lt;/span&gt; is
the method of choice, on real-world &lt;span class=SpellE&gt;datasets&lt;/span&gt; it is
almost uniformly worse than model trees.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;The comparison with linear regression depends on the error measure: for
one measure naïve &lt;span class=SpellE&gt;Bayes&lt;/span&gt; performs similarly, for
another it is worse.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Compared to
instance-based learning, it performs similarly with respect to both
measures.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;These results indicate that
the simplistic statistical assumption that naïve &lt;span class=SpellE&gt;Bayes&lt;/span&gt;
makes is indeed more restrictive for regression than for classification.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/16&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Link as you
type: using key phrases for automated dynamic link generation&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Steve Jones&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;When documents
are collected together from diverse sources they are unlikely to contain useful
hypertext links to support browsing amongst them.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;For large collections of thousands of
documents it is prohibitively resource intensive to manually insert links into
each document.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Users of such collections
may wish to relate documents within them to text that they are themselves
generating.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;This process, often
involving keyword searching, distracts from the authoring process and results
in material related to query terms but not necessarily to the author’s
document.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Query terms that are effective
in one collection might not be so in another.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;We have developed &lt;span class=SpellE&gt;Phrasier&lt;/span&gt;, a system that
integrates authoring (of text and hyperlinks), browsing, querying and reading
in support of information retrieval activities.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;&lt;span class=SpellE&gt;Phrasier&lt;/span&gt; exploits key phrases which are
automatically extracted from documents in a collection, and uses them as link
anchors and to identify candidate destinations for hyperlinks.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;This system suggests links into existing
collections for purposes of authoring and retrieval of related information,
creates links between documents in a collection and provides supportive
document and link overviews.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/17&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Melody based
tune retrieval over the World Wide Web&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;David
Bainbridge, &lt;span class=SpellE&gt;Rodger&lt;/span&gt; J. &lt;span class=SpellE&gt;McNab&lt;/span&gt;,
Lloyd A. Smith&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;In this paper we
describe the steps taken to develop a Web-based version of an existing
stand-alone, single-user digital library application for &lt;span class=SpellE&gt;melodical&lt;/span&gt;
searching of a collection of music.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;For
the three key components: input, searching, and output, we assess the
suitability of various Web-based strategies that deal with the now distributed
software architecture and explain the decisions we made.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;The resulting melody indexing service, known
as MELDEX, has been in operation for one year, and the feed-back we have
received has been &lt;span class=SpellE&gt;favorable&lt;/span&gt;.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;98/18&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Making oral
history accessible over the World Wide Web&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;David
Bainbridge, Sally Jo Cunningham&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;We describe a
multimedia, WWW-based oral history collection constructed from off-the-shelf or
publicly available software.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The source
materials for the collection include audio tapes of interviews and summary
transcripts of each interview, as well as photographs illustrating episodes
mentioned in the tapes.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Sections of the
transcripts are manually matched to associated segments of the tapes, and the
tapes are &lt;span class=SpellE&gt;digitized&lt;/span&gt;.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Users search a full-text retrieval system based on the text transcripts
to retrieve relevant transcript sections and their associated audio recordings
and photographs.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;It is also possible to
search for photos by matching text queries against text descriptions of the
photos in the collection, where the located photos link back to their
respective interview transcript and audio recordings.&lt;/span&gt;&lt;/p&gt;









&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;b style='mso-bidi-font-weight:
normal'&gt;&lt;span lang=EN-GB&gt;1997&lt;o:p&gt;&lt;/o:p&gt;&lt;/span&gt;&lt;/b&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/1&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;A dynamic and
flexible representation of social relationships in CSCW&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Steve Jones,
Steve Marsh&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;CSCW system
designers lack effective support in addressing the social issues and
interpersonal relationships which are linked with the use of CSCW systems.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;We present a formal description of trust to
support CSCW system designers in considering the social aspects of group work,
embedding those considerations in systems and analysing computer supported
group processes.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;We argue that
trust is a critical aspect in group work, and describe what we consider to be
the building blocks of trust.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We then
present a formal notation for the building blocks, their use in reasoning about
social interactions and how they are amended over time.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;We then consider
how the formalism may be used in practice, and present some insights from
initial analysis of the behaviour of the formalism.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;This is followed by a description of possible
amendments and extensions to the formalism.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;We conclude that it is possible to formalise a notion of trust and to
model the formalisation by a computational mechanism.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/2&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Design issues
for World Wide Web navigation visualisation tools&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Andy &lt;span
class=SpellE&gt;Cockburn&lt;/span&gt;, Steve Jones&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;The World Wide
Web (WWW) is a successful hypermedia information space used by millions of
people, yet it suffers from many deficiencies and problems in support for
navigation around its vast information space.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;In this paper we identify the origins of these navigation problems,
namely WWW browser design, WWW page design, and WWW page description
languages.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Regardless of their origins,
these problems are eventually represented to the user at the browser’s user
interface.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;To help overcome these
problems, many tools are being developed which allow users to visualise WWW
subspaces.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We identify five key issues
in the design and functionality of these visualisation systems: characteristics
of the visual representation, the scope of the subspace representation, the
mechanisms for generating the visualisation, the degree of browser
independence, and the navigation support facilities.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;We provide a critical review of the diverse
range of WWW visualisation tools with respect to these issues.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/3&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Stacked &lt;span
class=SpellE&gt;generalization&lt;/span&gt;:&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;when
does it work?&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Kai &lt;span
class=SpellE&gt;Ming&lt;/span&gt; Ting, Ian H. &lt;span class=SpellE&gt;Witten&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Stacked &lt;span
class=SpellE&gt;generalization&lt;/span&gt; is a general method of using a high-level
model to combine lower-level models to achieve greater predictive
accuracy.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;In this paper we address two
crucial issues which have been considered to be a 'black art' in classification
tasks ever since the introduction of stacked &lt;span class=SpellE&gt;generalization&lt;/span&gt;
in 1992 by &lt;span class=SpellE&gt;Wolpert&lt;/span&gt;: the type of &lt;span class=SpellE&gt;generalizer&lt;/span&gt;
that is suitable to derive the higher-level model, and the kind of attributes
that should be used as its input. &lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;We demonstrate
the effectiveness of stacked &lt;span class=SpellE&gt;generalization&lt;/span&gt; for
combining three different types of learning algorithms, and also for combining
models of the same type derived from a single learning algorithm in a
multiple-data-batches scenario.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We also
compare the performance of stacked &lt;span class=SpellE&gt;generalization&lt;/span&gt;
with published results arcing and bagging.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/4&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Browsing in
digital libraries:&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;a phrase-based
approach&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Craig &lt;span
class=SpellE&gt;Nevill&lt;/span&gt;-Manning, Ian H. &lt;span class=SpellE&gt;Witten&lt;/span&gt;,
Gordon W. &lt;span class=SpellE&gt;Paynter&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;A key question
for digital libraries is this: how should one go about becoming familiar with a
digital collection, as opposed to a physical one?&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Digital collections generally present an
appearance which is extremely opaque-a screen, typically a Web page, with no
indication of what, or how much, lies beyond: whether a carefully-selected
collection or a morass of worthless ephemera; whether half a dozen documents or
many millions.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;At least physical
collections occupy physical space, present a physical appearance, and exhibit
tangible physical &lt;span class=SpellE&gt;organization&lt;/span&gt;.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;When standing on the threshold of a large
library one gains a sense of presence and permanence that reflects the care
taken in building and maintaining the collection inside.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;No-one could confuse it with a
dung-heap!&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Yet in the digital world the
difference is not so palpable.&lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/5&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;A graphical
notation for the design of information visualisations&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Matthew C. &lt;span
class=SpellE&gt;Humphrey&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Visualisations
are coherent, graphical expressions of complex information that enhance people’s
ability to communicate and reason about that information.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Yet despite the importance of visualisations
in helping people to understand and solve a wide variety of problems, there is
a dearth of formal tools and methods for discussing, describing and designing
them.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Although simple visualisations,
such as bar charts and &lt;span class=SpellE&gt;scatterplots&lt;/span&gt;, are easily
produced by modern interactive software, novel visualisations of multivariate, &lt;span
class=SpellE&gt;multirelational&lt;/span&gt; data must be expressed in a programming
language.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The Relational Visualisation
Notation is a new, graphical language for designing such highly expressive
visualisations that does not use programming constructs.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Instead, the notation is based on relational
algebra, which is widely used in database query languages, and it is supported
by a suite of direct manipulation tools.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;This article presents the notation and examines the designs of some
interesting visualisations.&lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/6&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Applications of
machine learning in information retrieval&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Sally Jo
Cunningham, James &lt;span class=SpellE&gt;Littin&lt;/span&gt;, Ian H. &lt;span class=SpellE&gt;Witten&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Information
retrieval systems provide access to collections of thousands, or millions, of
documents, from which, by providing an appropriate description, users can
recover any one.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Typically, users &lt;span
class=SpellE&gt;iteratively&lt;/span&gt; refine the descriptions they provide to satisfy
their needs, and retrieval systems can &lt;span class=SpellE&gt;utilize&lt;/span&gt; user
feedback on selected documents to indicate the accuracy of the description at
any stage.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The style of description
required from the user, and the way it is employed to search the document
database, are consequences of the indexing method used for the collection.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;The index may take different forms, from
storing keywords with links to individual documents, to clustering documents
under related topics.&lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/7&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Computer
concepts without computers:&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;a first
course in computer science&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Geoffrey Holmes,
Tony C. Smith, William J. Rogers&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;While some
institutions seek to make CS1 curricula more enjoyable by incorporating
specialised educational software [1] or by setting more enjoyable programming
assignments [2], we have joined the growing number of Computer Science
departments that seek to improve the quality of the CS1 experience by focusing
student attention away from the computer monitor [3,4].&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Sophisticated computing concepts usually
reserved for senior level courses are presented in a &amp;lt;I&amp;gt;popular
science&amp;lt;/I&amp;gt; manner, and given equal time alongside the essential
introductory programming material.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;By
exposing students to a broad range of specific computational problems we
endeavour to make the introductory course more interesting and enjoyable, and
instil in students a sense of vision for areas they might specialise in as
computing majors.&lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/8&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;A sight-singing
tutor&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Lloyd A. Smith, &lt;span
class=SpellE&gt;Rodger&lt;/span&gt; J. &lt;span class=SpellE&gt;McNab&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;This paper
describes a computer program designed to aid its users in learning to
sight-sing.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Sight-singing-the ability to
sing music from a score without prior study-is an important skill for musicians
and holds a central place in most university music curricula.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Its importance to vocalists is obvious; it is
also an important skill for instrumentalists and conductors because it develops
the aural imagination necessary to judge how the music should sound, when
played (&lt;span class=SpellE&gt;Benward&lt;/span&gt; and Carr 1991).&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Furthermore, it is an important skill for
amateur musicians, who can save a great deal of rehearsal time through an
ability to sing music at sight.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/9&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Stacking bagged
and &lt;span class=SpellE&gt;dagged&lt;/span&gt; models&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Kai &lt;span
class=SpellE&gt;Ming&lt;/span&gt; Ting, I.H. &lt;span class=SpellE&gt;Witten&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;In this paper,
we investigate the method of &lt;i style='mso-bidi-font-style:normal'&gt;stacked &lt;span
class=SpellE&gt;generalization&lt;/span&gt;&lt;/i&gt; in combining models derived from
different subsets of a training &lt;span class=SpellE&gt;dataset&lt;/span&gt; by a single
learning algorithm, as well as different algorithms.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;The simplest way to combine predictions from
competing models is majority vote, and the effect of the sampling regime used
to generate training subsets has already been studied in this context-when
bootstrap samples are used the method is called &lt;i style='mso-bidi-font-style:
normal'&gt;bagging&lt;/i&gt;, and for disjoint samples we call it &lt;span class=SpellE&gt;&lt;i
style='mso-bidi-font-style:normal'&gt;dagging&lt;/i&gt;&lt;/span&gt;.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;This paper extends these studies to stacked &lt;span
class=SpellE&gt;generalization&lt;/span&gt;, where a learning algorithm is employed to combine
the models.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;This yields new methods
dubbed &lt;i style='mso-bidi-font-style:normal'&gt;bag-stacking&lt;/i&gt; and &lt;span
class=SpellE&gt;&lt;i style='mso-bidi-font-style:normal'&gt;dag&lt;/i&gt;&lt;/span&gt;&lt;i
style='mso-bidi-font-style:normal'&gt;-stacking&lt;/i&gt;.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;We demonstrate
that bag-stacking and &lt;span class=SpellE&gt;dag&lt;/span&gt;-stacking can be effective
for classification tasks even when the training samples cover just a small
fraction of the full &lt;span class=SpellE&gt;dataset&lt;/span&gt;.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;In contrast to earlier bagging results, we
show that bagging and bag-stacking work for stable as well as unstable learning
algorithms, as do &lt;span class=SpellE&gt;dagging&lt;/span&gt; and &lt;span class=SpellE&gt;dag&lt;/span&gt;-stacking.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;We find that bag-stacking (&lt;span
class=SpellE&gt;dag&lt;/span&gt;-stacking) almost always has higher predictive accuracy
than bagging (&lt;span class=SpellE&gt;dagging&lt;/span&gt;), and we also show that
bag-stacking models derived using two different algorithms is more effective
than bagging.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/10&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Extracting text
from Postscript&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Craig &lt;span
class=SpellE&gt;Nevill&lt;/span&gt;-Manning, Todd Reed, Ian H. &lt;span class=SpellE&gt;Witten&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;We show how to
extract plain text from PostScript files. A textual scan is inadequate because
PostScript interpreters can generate characters on the page that do not appear
in the source file. Furthermore, word and line breaks are implicit in the
graphical rendition, and must be inferred from the positioning of word
fragments. We present a robust technique for extracting text and &lt;span
class=SpellE&gt;recognizing&lt;/span&gt; words and paragraphs. The method uses a
standard PostScript interpreter but redefines several PostScript operators, and
simple heuristics are employed to locate word and line breaks. The scheme has
been used to create a full-text index, and plain-text versions, of 40,000
technical reports (34 &lt;span class=SpellE&gt;Gbyte&lt;/span&gt; of PostScript). Other
text-extraction systems are reviewed: none offer the same combination of
robustness and simplicity.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/11&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Gathering and
indexing rich fragments of the World Wide Web&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Geoffrey Holmes,
William J Rogers&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;While the World
Wide Web (WWW) is an attractive option as a resource for teaching and research
it does have some undesirable features. The cost of allowing students unlimited
access can be high-both in money and time; students may become addicted to
'surfing' the web-exploring purely for entertainment-and jeopardise their
studies. Students are likely to discover undesirable material because large
scale search engines index sites regardless of their merit. Finally, the
explosive growth of WWW usage means that servers and networks are often
overloaded, to the extent that a student may gain a very negative view of the
technology.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;We have developed
a piece of software which attempts to address these issues by capturing rich
fragments of the WWW onto local storage media. It is possible to put a
collection onto CD ROM, providing portability and inexpensive storage. This
enables the presentation of the WWW to distance learning students, who do not
have internet access. The software interfaces to standard, commonly available
web browsers, acting as a proxy server to the files stored on the local media,
and provides a search engine giving full text searching capability within the
collection.&lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/12&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Using model
trees for classification&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span class=SpellE&gt;&lt;span
lang=EN-GB&gt;Eibe&lt;/span&gt;&lt;/span&gt;&lt;span lang=EN-GB&gt; Frank, Yong Wang, Stuart &lt;span
class=SpellE&gt;Inglis&lt;/span&gt;, Geoffrey Holmes, Ian H. &lt;span class=SpellE&gt;Witten&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Model trees,
which are a type of decision tree with linear regression functions at the
leaves, form the basis of a recent successful technique for predicting
continuous numeric values.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;They can be
applied to classification problems by employing a standard method of
transforming a classification problem into a problem of function
approximation.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Surprisingly, using this
simple transformation the model tree &lt;span class=SpellE&gt;inducer&lt;/span&gt; M5',
based on &lt;span class=SpellE&gt;Quinlan's&lt;/span&gt; M5, generates more accurate
classifiers than the state-of-the-art decision tree learner C5.0, particularly
when most of the attributes are numeric.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/13&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Discovering inter-attribute
relationships&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Geoffrey Holmes&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;It is important
to discover relationships between attributes being used to predict a class
attribute in supervised learning situations for two reasons.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;First, any such relationship will be
potentially interesting to the provider of a &lt;span class=SpellE&gt;dataset&lt;/span&gt;
in its own right.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Second, it would
simplify a learning algorithm's search space, and the related irrelevant
feature and subset selection problem, if the relationships were removed from &lt;span
class=SpellE&gt;datasets&lt;/span&gt; ahead of learning.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;An algorithm to discover such relationships is presented in this
paper.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The algorithm is described and a
surprising number of inter-attribute relationships are discovered in &lt;span
class=SpellE&gt;datasets&lt;/span&gt; from the University of California at Irvine (UCI)
repository.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/14&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Learning from &lt;span
class=SpellE&gt;batched&lt;/span&gt; data:&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;model
combination &lt;span class=SpellE&gt;vs&lt;/span&gt; data combination&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Kai &lt;span
class=SpellE&gt;Ming&lt;/span&gt; Ting, Boon &lt;span class=SpellE&gt;Toh&lt;/span&gt; Low, Ian H. &lt;span
class=SpellE&gt;Witten&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;When presented
with multiple batches of data, one can either combine them into a single batch
before applying a machine learning procedure or learn from each batch
independently and combine the resulting models.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;The former procedure, data combination, is straightforward; this paper
investigates the latter, model combination.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Given an appropriate combination method, one might expect model
combination to prove superior when the data in each batch was obtained under
somewhat different conditions or when different learning algorithms were used
on the batches.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Empirical results show
that model combination often outperforms data combination even when the batches
are drawn randomly from a single source of data and the same learning method is
used on each.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Moreover, this is not just
an &lt;span class=SpellE&gt;artifact&lt;/span&gt; of one particular method of combining
models: it occurs with several different combination methods.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;We relate this
phenomenon to the learning curve of the classifiers being used.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Early in the learning process when the
learning curve is steep there is much to gain from data combination, but later
when it becomes shallow there is less to gain and model combination achieves a
greater reduction in variance and hence a lower error rate.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;The practical
implication of these results is that one should consider using model
combination rather than data combination, especially when multiple batches of
data for the same task are readily available.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;It is often superior even when the batches are drawn randomly from a
single sample, and we expect its advantage to increase if genuine statistical
differences between the batches exist.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/15&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Information
seeking retrieval, reading and storing behaviour of library users&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Turner K.&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;In the interest
of digital libraries, it is advisable that designers be aware of the potential
behaviour of the users of such a system.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;There are two distinct parts under investigation, the interaction
between traditional libraries involving the seeking and retrieval of relevant
material, and the reading and storage behaviours ensuing. Through this
analysis, the findings could be incorporated into digital library facilities.
There has been copious amounts of research on information seeking leading to
the development of behavioural models to describe the process. Often research
on the information seeking practices of individuals is based on the task and
field of study. The information seeking model, presented by Ellis et al.
(1993), characterises the format of this study where it is used to compare
various research on the information seeking practices of groups of people (from
academics to professionals). It is found that, although researchers do make use
of library facilities, they tend to rely heavily on their own collections and
primarily use the library as a source for previously identified information,
browsing and &lt;span class=SpellE&gt;interloan&lt;/span&gt;. It was found that there are
significant differences in user behaviour between the groups analysed. When
looking at the reading and storage of material it was hard to draw conclusions,
due to the lack of substantial research and information on the topic. However,
through the use of reading strategies, a general idea on how readers behave can
be developed. Designers of digital libraries can benefit from the guidelines
presented here to better understand their audience.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/16&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Proceeding of
the INTERACT97 Combined Workshop on CSCW in HCI-&lt;span class=SpellE&gt;Worldwide&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Matthias &lt;span
class=SpellE&gt;Rauterberg&lt;/span&gt;, Lars &lt;span class=SpellE&gt;Oestreicher&lt;/span&gt;,
John &lt;span class=SpellE&gt;Grundy&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;This is the
proceedings for the INTERACT97 combined workshop on “CSCW in HCI-&lt;span
class=SpellE&gt;worldwide&lt;/span&gt;”.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The
position papers in this proceedings are those selected from topics relating to
HCI community development &lt;span class=SpellE&gt;worldwide&lt;/span&gt; and to CSCW
issues.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Originally these were to be two
separate INTERACT workshops, but were combined to ensure sufficient
participation for a combined workshop to run.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;The combined
workshop has been split into two separate sessions to run in the morning of
July 15&lt;sup&gt;th&lt;/sup&gt;, Sydney, Australia.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;One to discuss the issues relating to the position papers focusing on
general CSCW systems, the other to the development of HCI communities in a &lt;span
class=SpellE&gt;worldwide&lt;/span&gt; context.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;The CSCW session uses as a case study a proposed &lt;span class=SpellE&gt;groupware&lt;/span&gt;
tool for facilitating the development of an HCI database with a &lt;span
class=SpellE&gt;worldwide&lt;/span&gt; geographical distribution.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;The HCI community session focuses on
developing the content for such a database, in order for it to foster the
continued development of HCI communities.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;The afternoon session of the combined workshop involves a joint
discussion of the case study &lt;span class=SpellE&gt;groupware&lt;/span&gt; tool, in terms
of its content and likely &lt;span class=SpellE&gt;groupware&lt;/span&gt; facilities.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;The position
papers have been grouped into those focusing on HCI communities and hence
content issues for a &lt;span class=SpellE&gt;groupware&lt;/span&gt; database, and those focusing
on CSCW and &lt;span class=SpellE&gt;groupware&lt;/span&gt; issues, and hence likely &lt;span
class=SpellE&gt;groupware&lt;/span&gt; support in the proposed HCI
database/collaboration tools.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We hope
that you find the position papers in this proceedings offer a wide range of
interesting reports of HCI community development &lt;span class=SpellE&gt;worldwide&lt;/span&gt;,
leading CSCW system research, and that a &lt;span class=SpellE&gt;groupware&lt;/span&gt;
tool supporting aspects of a &lt;span class=SpellE&gt;worldwide&lt;/span&gt; HCI database
can draw upon the varied work reported.&lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/17&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Internationalising
a spreadsheet for Pacific Basin languages&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Robert &lt;span
class=SpellE&gt;Barbour&lt;/span&gt;, Alvin &lt;span class=SpellE&gt;Yeo&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;As people trade
and engage in commerce, an economically dominant culture tends to migrate
language into other recently contacted cultures.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Information technology (IT) can accelerate &lt;span
class=SpellE&gt;enculturation&lt;/span&gt; and promote the expansion of western hegemony
in IT.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Equally, IT can present a
culturally appropriate interface to the user that promotes the preservation of
culture and language with very little additional effort.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;In this paper a spreadsheet is
internationalised to accept languages from the Latin-1 character set such as
English, &lt;span class=SpellE&gt;Maori&lt;/span&gt; and &lt;span class=SpellE&gt;Bahasa&lt;/span&gt; &lt;span
class=SpellE&gt;Melayu&lt;/span&gt; (Malaysia’s national language).&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;A technique that allows a non-programmer to
add a new language to the spreadsheet is described.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;The technique could also be used to
internationalise other software at the point of design by following the steps
we outline.&lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/18&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Localising a
spreadsheet:&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;an &lt;span class=SpellE&gt;Iban&lt;/span&gt;
example&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Alvin &lt;span
class=SpellE&gt;Yeo&lt;/span&gt;, Robert &lt;span class=SpellE&gt;Barbour&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Presently, there
is little localisation of software to smaller cultures if it is not
economically viable.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We believe software
should also be localised to the languages of small cultures in order to sustain
and preserve these small cultures.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;As an
example, we localised a spreadsheet from English to &lt;span class=SpellE&gt;Iban&lt;/span&gt;.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;The process in which we carried out the
localisation can be used as a framework for the localisation of software to
languages of small ethnic minorities.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Some problems faced during the localisation process are also discussed.&lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/19&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Strategies of
internationalisation and localisation: a postmodernist/s perspective&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Alvin &lt;span
class=SpellE&gt;Yeo&lt;/span&gt;, Robert &lt;span class=SpellE&gt;Barbour&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Many software
companies today are developing software not only for local consumption but for
the rest of the world.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We introduce the
concepts of internationalisation and localisation and discuss some techniques
using these processes.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;An examination of
&lt;span class=SpellE&gt;postmodern&lt;/span&gt; critique with respect to the software
industry is also reported.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;In addition,
we also feature our proposed internationalisation technique that was inspired
by taking into account the researches of &lt;span class=SpellE&gt;postmodern&lt;/span&gt;
philosophers and mathematicians.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;As illustrated
in our prototype, the technique empowers non-programmers to localise their own
software.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Further development of the
technique and its implications on user interfaces and the future of software
internationalisation and localisation are discussed.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/20&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Language use in
software&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Alvin &lt;span
class=SpellE&gt;Yeo&lt;/span&gt;, Robert &lt;span class=SpellE&gt;Barbour&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Many of the
popular software we use today are in English.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Very few software applications are available in minority languages.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Besides economic goals, we justify why
software should be made available to smaller cultures.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Furthermore, there is evidence that people
learn and progress faster in software in their mother tongue (&lt;span
class=SpellE&gt;Griffiths&lt;/span&gt; et at, 1994) (&lt;span class=SpellE&gt;Krock&lt;/span&gt;,
1996).&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We hypothesise that experienced
users of English spreadsheet can easily migrate to a spreadsheet in their
native tongue i.e. &lt;span class=SpellE&gt;Bahasa&lt;/span&gt; &lt;span class=SpellE&gt;Melayu&lt;/span&gt;
(Malaysia’s national language).&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Observations made in the study suggest that the native speakers of &lt;span
class=SpellE&gt;Bahasa&lt;/span&gt; &lt;span class=SpellE&gt;Melayu&lt;/span&gt; had difficulties
with the &lt;span class=SpellE&gt;Bahasa&lt;/span&gt; &lt;span class=SpellE&gt;Melayu&lt;/span&gt;
interface.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The subjects’ main difficulty
was their unfamiliarity with computing terminology in &lt;span class=SpellE&gt;Bahasa&lt;/span&gt;
&lt;span class=SpellE&gt;Melayu&lt;/span&gt;.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We
present possible strategies to increase the use of &lt;span class=SpellE&gt;Bahasa&lt;/span&gt;
&lt;span class=SpellE&gt;Melayu&lt;/span&gt; in IT.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;These strategies may also be used to promote the use of other minority
languages in IT.&lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/21&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Usability
testing:&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;a Malaysian study&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Alvin &lt;span
class=SpellE&gt;Yeo&lt;/span&gt;, Robert &lt;span class=SpellE&gt;Barbour&lt;/span&gt;, Mark &lt;span
class=SpellE&gt;Apperley&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;An exploratory
study of software assessment techniques is conducted in Malaysia.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Subjects in the study comprised staff members
of a Malaysian university with a high Information Technology (IT) presence.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;The subjects assessed a spreadsheet tool with
a &lt;span class=SpellE&gt;Bahasa&lt;/span&gt; &lt;span class=SpellE&gt;Melayu&lt;/span&gt; (Malaysia’s
national language) interface.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Software
evaluation techniques used include the think aloud method, interviews and the
System Usability Scale.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The responses in
the various techniques used are reported and initial results indicate
idiosyncratic behaviour of Malaysian subjects.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;The implications of the findings are also discussed.&lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/22&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Inducing
cost-sensitive trees via instance-weighting&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Kai &lt;span
class=SpellE&gt;Ming&lt;/span&gt; Ting&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;We introduce an
instance-weighting method to induce cost-sensitive trees in this paper.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;It is a &lt;span class=SpellE&gt;generalization&lt;/span&gt;
of the standard tree induction process where only the initial instance weights
determine the type of tree (i.e., minimum error trees or minimum cost trees) to
be induced.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We demonstrate that it can
be easily adopted to an existing tree learning algorithm.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Previous
research gave insufficient evidence to support the fact that the greedy
divide-and-conquer algorithm can effectively induce a truly cost-sensitive tree
directly from the training data.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We
provide this empirical evidence in this paper.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;The algorithm employing the instance-weighting method is found to be
comparable to or better than both C4.5 and C5 in terms of total
misclassification costs, tree size and the number of high cost errors.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;The instance-weighting method is also simpler
and more effective in implementation than a method based on altered priors.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/23&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Fast convergence
with a greedy tag-phrase dictionary&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Ross &lt;span
class=SpellE&gt;Peeters&lt;/span&gt;, Tony C. Smith&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoBodyText&gt;&lt;span lang=EN-US&gt;The best general-purpose compression
schemes make their gains by estimating a probability distribution over all
possible next symbols given the context established by some number of previous
symbols.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Such context models typically
obtain good compression results for plain text by taking advantage of
regularities in character sequences.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Frequent words and syllables can be incorporated into the model quickly
and thereafter used for reasonably accurate prediction.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;However, the precise context in which
frequent patterns emerge is often extremely varied, and each new word or phrase
immediately introduces new contexts which can adversely affect the compression
rate&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;A great deal of
the structural regularity in a natural language is given rather more by
properties of its grammar than by the orthographic transcription of its
phonology.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;This implies that access to a
grammatical abstraction might lead to good compression.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;While grammatical models have been used
successfully for compressing computer programs [4], grammar-based compression
of plain text has received little attention, primarily because of the
difficulties associated with constructing a suitable natural language
grammar.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;But even without a precise
formulation of the syntax of a language, there is a linguistic abstraction
which is easily accessed and which demonstrates a high degree of regularity
which can be exploited for compression purposes-namely, lexical categories.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/24&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Tag based models
of English text&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;W. J. &lt;span
class=SpellE&gt;Teahan&lt;/span&gt;, John G. &lt;span class=SpellE&gt;Cleary&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;The problem of
compressing English text is important both because of the ubiquity of English
as a target for compression and because of the light that compression can shed
on the structure of English.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;English
text is examined in conjunction with additional information about the parts of
speech of each word in the text (these are referred to as “tags”).&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;It is shown that the tags plus the text can
be compressed more than the text alone.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;Essentially the tags can be compressed for nothing or even a small net
saving in size.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;A comparison is made of
a number of different ways of integrating compression of tags and text using an
escape mechanism similar to PPM.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;These
are also compared with standard word based and character based compression
programs.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The result is that the tag
character and word based schemes always outperform the character based
schemes.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Overall, the tag based schemes
outperform the word based schemes.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We
conclude by conjecturing that tags chosen for compression rather than
linguistic purposes would perform even better.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/25&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Musical image
compression&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;David
Bainbridge, Stuart &lt;span class=SpellE&gt;Inglis&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Optical music
recognition aims to convert the vast repositories of sheet music in the world
into an on-line digital format [Bai97].&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;In the near future it will be possible to assimilate music into digital
libraries and users will be able to perform searches based on a sung melody in
addition to typical text-based searching [MSW+96].&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;An important requirement for such a system is
the ability to reproduce the original score as accurately as possible.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Due to the huge amount of sheet music
available, the efficient storage of musical images is an important topic of
study.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;This paper
investigates whether the “knowledge” extracted from the optical music
recognition (OMR) process can be exploited to gain higher compression than the
JBIG international standard for &lt;span class=SpellE&gt;bi&lt;/span&gt;-level image
compression.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We present a hybrid
approach where the primitive shapes of music extracted by the optical music
recognition process-note heads, note stems, staff lines and so forth-are fed
into a graphical symbol based compression scheme originally designed for images
containing mainly printed text.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Using
this hybrid approach the average compression rate for a single page is improved
by 3.5% over JBIG.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;When multiple pages with
similar typography are processed in sequence, the file size is decreased by
4-8%.&lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Section 2
presents the relevant background to both optical music recognition and textual
image compression.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Section 3 describes
the experiments performed on 66 test images, outlining the combinations of
parameters that were examined to give the best results.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;The initial results and refinements are
presented in Section 4, and we conclude in the last section by &lt;span
class=SpellE&gt;summarizing&lt;/span&gt; the findings of this work.&lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/26&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Correcting English
text using PPM models&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;W. J. &lt;span
class=SpellE&gt;Teahan&lt;/span&gt;, S. &lt;span class=SpellE&gt;Inglis&lt;/span&gt;, J. G. &lt;span
class=SpellE&gt;Cleary&lt;/span&gt;, G. Holmes&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;An essential
component of many applications in natural language processing is a language &lt;span
class=SpellE&gt;modeler&lt;/span&gt; able to correct errors in the text being
processed.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;For optical character recognition
(OCR), poor scanning quality or extraneous pixels in the image may cause one or
more characters to be mis-&lt;span class=SpellE&gt;recognized&lt;/span&gt;; while for
spelling correction, two characters may be transposed, or a character may be
inadvertently inserted or missed out. &lt;/span&gt;&lt;/p&gt;



&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;This paper
describes a method for correcting English text using a PPM model.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;A method that segments words in English text
is introduced and is shown to be a significant improvement over previously used
methods.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;A similar technique is also
applied as a post-processing stage after pages have been &lt;span class=SpellE&gt;recognized&lt;/span&gt;
by a state-of-the-art commercial OCR system.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;We show that the accuracy of the OCR system can be increased from 95.9%
to 96.6%, a decrease of about 10 errors per page.&lt;/span&gt;&lt;/p&gt;









&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/27&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Constraints on
parallelism beyond 10 instructions per cycle&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;John G. &lt;span
class=SpellE&gt;Cleary&lt;/span&gt;, Richard H. &lt;span class=SpellE&gt;Littin&lt;/span&gt;, J. A.
David &lt;span class=SpellE&gt;McWha&lt;/span&gt;, Murray W. Pearson&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;The problem of
extracting Instruction Level Parallelism at levels of 10 instructions per clock
and higher is considered.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Two different
architectures which use speculation on memory accesses to achieve this level of
performance are reviewed.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;It is pointed
out that while this form of speculation gives high potential parallelism it is
necessary to retain execution state so that incorrect speculation can be detected
and subsequently squashed.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Simulation
results show that the space to store such state is a critical resource in
obtaining good speedup.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;To make good use
of the space it is essential that state be stored efficiently and that it be
retired as soon as possible.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;A number of
techniques for extracting the best usage from the available state storage are
introduced.&lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/28&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Effects of
re-ordered memory operations on parallelism&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;Richard H. &lt;span
class=SpellE&gt;Littin&lt;/span&gt;, John G. &lt;span class=SpellE&gt;Cleary&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;The performance
effect of permitting different memory operations to be re-ordered is
examined.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The available parallelism is
computed using a machine code simulator.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;A range of possible restrictions on the re-ordering of memory operations
is considered: from the purely sequential case where no re-ordering is
permitted; to the completely permissive one where memory operations may occur
in any order so that the parallelism is restricted only by data
dependencies.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;A general conclusion is
drawn that to reliably obtain parallelism beyond 10 instructions per clock will
require an ability to re-order all memory instructions.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;A brief description of a feasible
architecture capable of this is given.&lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal style='margin-right:-.4pt'&gt;&lt;span lang=EN-GB&gt;97/29&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal&gt;&lt;span lang=EN-GB&gt;OZCHI’96 Industry Session:&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Sixth Australian Conference on Human-Computer
Interaction&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal&gt;&lt;span lang=EN-GB&gt;Edited by Chris Phillips, Janis &lt;span
class=SpellE&gt;McKauge&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal&gt;&lt;span lang=EN-GB&gt;The idea for a specific industry session at
OZCHI was first mooted at the 1995 conference in &lt;span class=SpellE&gt;Wollongong&lt;/span&gt;,
during questions following a session of short papers which happened
(serendipitously) to be presented by people from industry.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;An animated discussion took place, most of
which was about how OZCHI could be made more relevant to people in industry, be
it working as usability consultants, or working within organisations either as
usability professionals or as ‘champions of the cause’.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;The discussion raised more questions than
answers, about the format of such as session, about the challenges of
attracting industry participation, and about the best way of publishing the
results.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;Although no real solutions were
arrived at, it was enough to place an industry session on the agenda for
OZCHI’96.&lt;/span&gt;&lt;/p&gt;





&lt;p class=MsoNormal&gt;&lt;span lang=EN-GB&gt;97/30&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal&gt;&lt;span lang=EN-GB&gt;Adaptive models of English text&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal&gt;&lt;span lang=EN-GB&gt;W. J. &lt;span class=SpellE&gt;Teahan&lt;/span&gt;,
John G. &lt;span class=SpellE&gt;Cleary&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal&gt;&lt;span lang=EN-GB&gt;High quality models of English text with
performance approaching that of humans is important for many applications
including spelling correction, speech recognition, OCR, and encryption.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;A number of different statistical models of
English are compared with each other and with previous estimates from human
subjects.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;It is concluded that the best
current models are word based with part of speech tags.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Given sufficient training text, they are able
to attain performance comparable to humans.&lt;/span&gt;&lt;/p&gt;







&lt;p class=MsoNormal&gt;&lt;span lang=EN-GB&gt;97/31&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal&gt;&lt;span lang=EN-GB&gt;A graphical user interface for Boolean
query specification&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal&gt;&lt;span lang=EN-GB&gt;Steve Jones, &lt;span class=SpellE&gt;Shona&lt;/span&gt;
&lt;span class=SpellE&gt;McInnes&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;

&lt;p class=MsoNormal&gt;&lt;span lang=EN-GB&gt;On-line information repositories commonly
provide keyword search facilities via textual query languages based on Boolean
logic.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;However, there is evidence to
suggest that the syntactical demands of such languages can lead to user errors
and adversely affect the time that it takes users to form queries.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;Users also face difficulties because of the
conflict in semantics between AND &lt;span class=SpellE&gt;and&lt;/span&gt; OR when used in
Boolean logic and English language.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We
suggest that graphical query languages, in particular Venn-like diagrams, can
alleviate the problems that users experience when forming Boolean expressions
with textual languages.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We describe &lt;span
class=SpellE&gt;Vquery&lt;/span&gt;, a Venn-diagram based user interface to the New
Zealand Digital Library (NZDL).&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;The
design of &lt;span class=SpellE&gt;Vquery&lt;/span&gt; has been partly motivated by
analysis of NZDL usage.&lt;span style='mso-spacerun:yes'&gt;  &lt;/span&gt;We found that
few queries contain more than three terms, use of the intersection operator
dominates and that query refinement is common.&lt;span style='mso-spacerun:yes'&gt; 
&lt;/span&gt;A study of the utility of Venn diagrams for query specification
indicates that with little or no training users can interpret and form
Venn-like diagrams which accurately correspond to Boolean expressions.&lt;span
style='mso-spacerun:yes'&gt;  &lt;/span&gt;The utility of &lt;span class=SpellE&gt;Vquery&lt;/span&gt;
is considered and directions for future work are proposed.&lt;/span&gt;&lt;/p&gt;





&lt;/div&gt;




</Content>
</Section>
</Archive>
