本文实例为大家分享了英文单词统计器php 实现,供大家参考,具体内容如下
程序开始运行, 按"浏览"钮选择一个英文文档, 再按"统计 statistics"钮, 即可得到按字母顺序列出的所有单词,及其出现的次数
用于测试的数据文档: data.txt
驱动程序:word.php
output.php 和 stringtokenizer.php 是 要求在同一个文件夹中的程序
1. words_statistics_php.png
2. word.php
<html>
<style>
td{
background-color:#cf6;
width:100px;
margin:5px;
}
</style>
<body>
<?php
/**
* 程序开始运行, 按"浏览"钮选择一个英文文档, 再按"统计"钮,
* 即可得到按字母顺序列出的所有单词,及其出现的次数
*
* 作者: 许同春 author tongchun xu
* @开源中国 open source, chna communiity
* 完成日期:2016年6月10日 completion date: 10 june, 2016
*/
require("stringtokenizer.php");
require("output.php");
if($_post['submit']){
if ($_files["file"]["error"] > 0)
echo "error: " . $_files["file"]["error"] . "<br />";
else {
$myfile = fopen($_files["file"]["tmp_name"], "r") or die("unable to open file!");
$str = fread($myfile,filesize($_files["file"]["tmp_name"]));
$delim = "?\,. /:!"()tnrf%";
$st = new stringtokenizer($str, $delim);
echo '找到字符串: '.$st->counttokens();
$list=new linkedlist();
while ($st->hasmoretokens()) {
$list->orderinsert($st->nexttoken());
}
$list->words_count();
$list->traversal();
fclose($myfile);
}
}
?>
<h2>英文文档单词统计 statistics on english words </h2>
<p>程序开始运行, 按"浏览"钮选择一个英文文档, 再按"统计 statistics"钮,
即可得到按字母顺序列出的所有单词,及其出现的次数 </p>
<form action="word.php" method="post"
enctype="multipart/form-data">
<label for="file">英文文档名 file name:</label>
<input type="file" name="file" id="file" />
<input type="submit" name="submit" value="统计 statistics" />
</form>
</body>
</html>
3. output.php
<meta charset="utf-8" />
<?
/**
* the class linkedlist allows an application to store strings in
* alphabetical order by calling orderinsert().
* 此处定义的 linkedlist 类,可以调用它的 方法 orderinsert(),来以字母
* 大小的顺序储存 英文字符串。
* 同时记录 英文单词出现的次数
* 作者: 许同春 author tongchun xu
* @开源中国 open source, china communiity
* 完成日期:2016年6月10日 completion date: 10 june, 2016
*/
class node{
public $data;
public $frequency;
public $next;
function __construct($data, $next = null, $frequency = 1){
$this->data = $data; //英文字符串
$this->next = $next; //指向后继结点的指针
$this->frequency=$frequency; //英文字符串出现的次数
}
}
class linkedlist{
private $head; //单链表的头结点,不存储数据
function __construct(){//单链表的构造方法
//头结点的数据为"傀儡", 不代表 任何数据
$this->head = new node("dummy 傀儡");
$this->first = null;
}
function isempty(){
return ($this->head->next == null);
}
/* orderinsert($data) 方法,
* 按给定字符串 $data 的大小, 将其安插到适当的位置,
* 以保证单链表中字符串的存储,始终是有序的。
*/
function orderinsert($data){
$p = new node($data);
if($this->isempty()){
$this->head->next = $p;
}
else {
$node= $this->find($data);
if(!$node){
$q = $this->head;
while($q->next != null && strcmp($data, $q->next->data)> 0 ){
$q = $q->next;
}
$p->next = $q->next;
$q->next = $p;
}else
$node->frequency++;
}
}
function insertlast($data){//将字符串插到单链表的尾部
$p = new node($data);
if($this->isempty()){
$this->head->next = $p;
}
else{
$q = $this->head->next;
while($q->next != null)
$q = $q->next;
$q->next = $p;
}
}
function find($value){//查询是否有给定的字符串
$q = $this->head->next;
while($q->next != null){
if(strcmp($q->data,$value)==0){
break;
}
$q = $q->next;
}
if ($q->data == $value)
return $q;
else
return null;
}
function traversal(){//遍历单链表
if(!$this->isempty()){
$p=$this->head->next;
echo "输出结果:<table><tr>";
echo "<td>".$p->data."<br>出现次数:".$p->frequency."</td>";
$n=1;
while($p->next != null){
$p=$p->next;
echo "<td>".$p->data."<br>出现次数:".$p->frequency."</td>";
$n++;
if ($n%11==0) echo "</tr><tr>";
}
echo "</tr></table>";
}else
echo "链表为空!";
}
function words_count(){
if($this->isempty())
echo "<br>没有储存字符串 <br>";
else{
$counter=0;
$p=$this->head->next;
while($p->next != null){
$p=$p->next;
$counter++;
};
echo "***共有单词 ".$counter." 个***";
}
}}
?>
4. stringtokenizer.php
<?php
/**
* the string tokenizer class allows an application to break a string into tokens.
*
* @author azeem michael
* @example the following is one example of the use of the tokenizer. the code:
* <code>
* <?php
* $str = "this is:@tn a test!";
* $delim = " !@:'tn "; // remove these chars
* $st = new stringtokenizer($str, $delim);
* echo 'total tokens: '.$st->counttokens().'<br/>';
* while ($st->hasmoretokens()) {
* echo $st->nexttoken() . '<br/>';
* }
* prints the following output:
* total tokens: 4
* this
* is
* a
* test
* ?>
* </code>
*/
class stringtokenizer {
/** @var string
*/
private $string;
/** @var string
*/
private $token;
/** @var string
*/
private $delim;
/**
* constructs a string tokenizer for the specified string.
* @param string $str string to tokenize
* @param string $delim the set of delimiters (the characters that separate tokens)
* specified at creation time, default to " nrt "
*/
public function __construct($str, $delim=" nrt ") {
$this->string = $str;
$this->delim = $delim;
$this->token = strtok($str, $delim);
}
/**
* destructor to prevent memory leaks
*/
public function __destruct() {
unset($this);
}
/**
* calculates the number of times that this tokenizer's nexttoken method can
* be called before it generates an exception
* @return int - number of tokens
*/
public function counttokens() {
$counter = 0;
while($this->hasmoretokens()) {
$counter++;
$this->nexttoken();
}
$this->token = strtok($this->string, $this->delim);
return $counter;
}
/**
* tests if there are more tokens available from this tokenizer's string. it
* does not move the internal pointer in any way. to move the internal pointer
* to the next element call nexttoken()
* @return boolean - true if has more tokens, false otherwise
*/
public function hasmoretokens() {
return ($this->token !== false);
}
/**
* returns the next token from this string tokenizer and advances the internal
* pointer by one.
* @return string - next element in the tokenized string
*/
public function nexttoken() {
$hold = $this->token; //hold current pointer value
$this->token = strtok($this->delim); //increment pointer
return $hold; //return current pointer value
}
}
?>
以上就是本文的全部内容,希望对大家的学习有所帮助,也希望大家多多支持硕编程。
【说明】:本文章由站长整理发布,文章内容不代表本站观点,如文中有侵权行为,请与本站客服联系(QQ:254677821)!