<?xml version="1.0" encoding="UTF-8"?>

<rdf:RDF
   xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"
   xmlns:rdfs="http://www.w3.org/2000/01/rdf-schema#"
   xmlns="http://purl.org/rss/1.0/"
   xmlns:dc="http://purl.org/dc/elements/1.1/"
   xmlns:prism="http://prismstandard.org/namespaces/1.2/basic/"
   xmlns:dcterms="http://purl.org/dc/terms/"

>
<channel rdf:about="http://www.citeulike.org/about">
<pubDate>Thu, 21 Aug 2008 15:17:24 BST</pubDate>


	<title>CiteULike: briordan Luo</title>
	<description>CiteULike: briordan Luo</description>


	<link>http://www.citeulike.org/user/briordan/author/Luo</link>
	<dc:publisher>CiteULike.org</dc:publisher>
	<dc:language>en-gb</dc:language>
	<dc:rights>Copyright &#169; 2004-2008 citeulike.org</dc:rights>
	<items>
    <rdf:Seq>
        <rdf:li rdf:resource="http://www.citeulike.org/user/briordan/article/2624700"/>

	</rdf:Seq>
	</items>
	</channel>


<item rdf:about="http://www.citeulike.org/user/briordan/article/2624700">
    <title>Text Clustering with Feature Selection by Using Statistical Data</title>
    <link>http://www.citeulike.org/user/briordan/article/2624700</link>
    <description>&lt;i&gt;Knowledge and Data Engineering, IEEE Transactions on, Vol. 20, No. 5. (2008), pp. 641-652.&lt;/i&gt;&lt;br /&gt;&lt;br /&gt;Feature selection is an important method for improving the efficiency and accuracy of text categorization algorithms by removing redundant and irrelevant terms from the corpus. In this paper, we propose a new supervised feature selection method, named CHIR, which is based on the Chi-square statistic and new statistical data that can measure the positive term-category dependency. We also propose a new text clustering algorithm TCFS, which stands for Text Clustering with Feature Selection. TCFS can incorporate CHIR to identify relevant features (i.e., terms) iteratively, and the clustering becomes a learning process. We compared TCFS and the k-means clustering algorithm in combination with different feature selection methods for various real data sets. Our experimental results show that TCFS with CHIR has better clustering accuracy in terms of the F-measure and the purity.</description>
    <dc:title>Text Clustering with Feature Selection by Using Statistical Data</dc:title>

    <dc:creator>Yanjun Li</dc:creator>
    <dc:creator>Congnan Luo</dc:creator>
    <dc:creator>Soon Chung</dc:creator>
    <dc:identifier>doi:10.1109/TKDE.2007.190740</dc:identifier>
    <dc:source>Knowledge and Data Engineering, IEEE Transactions on, Vol. 20, No. 5. (2008), pp. 641-652.</dc:source>
    <dc:date>2008-04-03T00:25:55-00:00</dc:date>
    <prism:publicationYear>2008</prism:publicationYear>
    <prism:publicationName>Knowledge and Data Engineering, IEEE Transactions on</prism:publicationName>
    <prism:volume>20</prism:volume>
    <prism:number>5</prism:number>
    <prism:startingPage>641</prism:startingPage>
    <prism:endingPage>652</prism:endingPage>
    <prism:category>computational-linguistics</prism:category>
    <prism:category>machine-learning</prism:category>
</item>



</rdf:RDF>

