2020美赛c题建模数据+代码+论文H
@@ -0,0 +1,6 @@
|
|||||||
|
{
|
||||||
|
"cells": [],
|
||||||
|
"metadata": {},
|
||||||
|
"nbformat": 4,
|
||||||
|
"nbformat_minor": 2
|
||||||
|
}
|
||||||
@@ -0,0 +1,108 @@
|
|||||||
|
review_date,star_rating,count
|
||||||
|
2015-01-02,2,1
|
||||||
|
2015-01-02,3,2
|
||||||
|
2015-01-02,4,1
|
||||||
|
2015-01-02,5,6
|
||||||
|
2015-01-03,1,4
|
||||||
|
2015-01-03,2,1
|
||||||
|
2015-01-03,3,3
|
||||||
|
2015-01-03,4,3
|
||||||
|
2015-01-03,5,9
|
||||||
|
2015-01-04,1,1
|
||||||
|
2015-01-04,2,2
|
||||||
|
2015-01-04,3,1
|
||||||
|
2015-01-04,4,3
|
||||||
|
2015-01-04,5,8
|
||||||
|
2015-01-05,1,4
|
||||||
|
2015-01-05,3,2
|
||||||
|
2015-01-05,4,6
|
||||||
|
2015-01-05,5,14
|
||||||
|
2015-01-06,2,1
|
||||||
|
2015-01-06,3,3
|
||||||
|
2015-01-06,4,6
|
||||||
|
2015-01-06,5,8
|
||||||
|
2015-01-07,1,1
|
||||||
|
2015-01-07,2,4
|
||||||
|
2015-01-07,3,1
|
||||||
|
2015-01-07,4,5
|
||||||
|
2015-01-07,5,8
|
||||||
|
2015-01-08,1,1
|
||||||
|
2015-01-08,3,2
|
||||||
|
2015-01-08,4,2
|
||||||
|
2015-01-08,5,10
|
||||||
|
2015-01-09,1,2
|
||||||
|
2015-01-09,2,2
|
||||||
|
2015-01-09,3,2
|
||||||
|
2015-01-09,4,3
|
||||||
|
2015-01-09,5,20
|
||||||
|
2015-01-10,1,1
|
||||||
|
2015-01-10,3,2
|
||||||
|
2015-01-10,4,1
|
||||||
|
2015-01-10,5,4
|
||||||
|
2015-01-11,1,1
|
||||||
|
2015-01-11,3,1
|
||||||
|
2015-01-11,4,3
|
||||||
|
2015-01-11,5,10
|
||||||
|
2015-01-12,1,3
|
||||||
|
2015-01-12,2,1
|
||||||
|
2015-01-12,3,1
|
||||||
|
2015-01-12,4,4
|
||||||
|
2015-01-12,5,9
|
||||||
|
2015-01-13,3,1
|
||||||
|
2015-01-13,4,2
|
||||||
|
2015-01-13,5,15
|
||||||
|
2015-01-14,1,1
|
||||||
|
2015-01-14,3,3
|
||||||
|
2015-01-14,5,9
|
||||||
|
2015-01-15,1,1
|
||||||
|
2015-01-15,3,1
|
||||||
|
2015-01-15,4,3
|
||||||
|
2015-01-15,5,8
|
||||||
|
2015-01-16,3,3
|
||||||
|
2015-01-16,4,8
|
||||||
|
2015-01-16,5,5
|
||||||
|
2015-01-17,4,3
|
||||||
|
2015-01-17,5,6
|
||||||
|
2015-01-18,1,3
|
||||||
|
2015-01-18,2,2
|
||||||
|
2015-01-18,3,1
|
||||||
|
2015-01-18,4,1
|
||||||
|
2015-01-18,5,14
|
||||||
|
2015-01-19,1,1
|
||||||
|
2015-01-19,4,1
|
||||||
|
2015-01-19,5,4
|
||||||
|
2015-01-20,1,1
|
||||||
|
2015-01-20,3,3
|
||||||
|
2015-01-20,4,5
|
||||||
|
2015-01-20,5,14
|
||||||
|
2015-01-21,1,2
|
||||||
|
2015-01-21,2,1
|
||||||
|
2015-01-21,3,2
|
||||||
|
2015-01-21,4,1
|
||||||
|
2015-01-21,5,14
|
||||||
|
2015-01-22,1,1
|
||||||
|
2015-01-22,5,10
|
||||||
|
2015-01-23,2,1
|
||||||
|
2015-01-23,3,1
|
||||||
|
2015-01-23,5,3
|
||||||
|
2015-01-24,3,2
|
||||||
|
2015-01-24,4,1
|
||||||
|
2015-01-24,5,5
|
||||||
|
2015-01-25,1,1
|
||||||
|
2015-01-25,2,1
|
||||||
|
2015-01-25,4,3
|
||||||
|
2015-01-26,2,2
|
||||||
|
2015-01-26,5,8
|
||||||
|
2015-01-27,5,9
|
||||||
|
2015-01-28,3,2
|
||||||
|
2015-01-28,4,1
|
||||||
|
2015-01-28,5,6
|
||||||
|
2015-01-29,4,3
|
||||||
|
2015-01-29,5,16
|
||||||
|
2015-01-30,1,1
|
||||||
|
2015-01-30,2,1
|
||||||
|
2015-01-30,5,5
|
||||||
|
2015-01-31,2,2
|
||||||
|
2015-01-31,3,2
|
||||||
|
2015-01-31,4,5
|
||||||
|
2015-01-31,5,8
|
||||||
|
|
After Width: | Height: | Size: 2.5 KiB |
|
After Width: | Height: | Size: 30 KiB |
@@ -0,0 +1,43 @@
|
|||||||
|
{
|
||||||
|
"cells": [
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": null,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"import requests\n",
|
||||||
|
"import random\n",
|
||||||
|
"user_agend=[\n",
|
||||||
|
" 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/73.0.3683.103 Safari/537.36',\n",
|
||||||
|
" 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_7_0) AppleWebKit/535.11 (KHTML, like Gecko) Chrome/17.0.963.56 Safari/535.11' \n",
|
||||||
|
" ]\n",
|
||||||
|
"\n",
|
||||||
|
"header={\n",
|
||||||
|
" \n",
|
||||||
|
"}"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"metadata": {
|
||||||
|
"kernelspec": {
|
||||||
|
"display_name": "Python 3",
|
||||||
|
"language": "python",
|
||||||
|
"name": "python3"
|
||||||
|
},
|
||||||
|
"language_info": {
|
||||||
|
"codemirror_mode": {
|
||||||
|
"name": "ipython",
|
||||||
|
"version": 3
|
||||||
|
},
|
||||||
|
"file_extension": ".py",
|
||||||
|
"mimetype": "text/x-python",
|
||||||
|
"name": "python",
|
||||||
|
"nbconvert_exporter": "python",
|
||||||
|
"pygments_lexer": "ipython3",
|
||||||
|
"version": "3.7.1"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"nbformat": 4,
|
||||||
|
"nbformat_minor": 2
|
||||||
|
}
|
||||||
|
After Width: | Height: | Size: 73 KiB |
|
After Width: | Height: | Size: 85 KiB |
@@ -0,0 +1,351 @@
|
|||||||
|
{
|
||||||
|
"cells": [
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## Q2"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 2,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"and daughter (V.W.).\n",
|
||||||
|
")[Gal(beta 1,3/4)Glc-NAc(beta 1,?)]\n",
|
||||||
|
"[5] is to jointly learn an em- bedding of words into an n-dimensional vector space and to use these vectors to predict how likely a word is given its context.\n",
|
||||||
|
".\n",
|
||||||
|
".\n",
|
||||||
|
", xm ).\n",
|
||||||
|
"(1992) J. Biol.\n",
|
||||||
|
"267, 968-974; Zhou et al.\n",
|
||||||
|
"(1995) Mol.\n",
|
||||||
|
"9, 208-218).\n",
|
||||||
|
"41, 115-123) largely failed to stain tubular basement membranes, suggesting the presence of heparan sulfate chains lacking the specific JM-403 epitope.\n",
|
||||||
|
"['and daughter (V.W.).', ')[Gal(beta 1,3/4)Glc-NAc(beta 1,?)]', '[5] is to jointly learn an em- bedding of words into an n-dimensional vector space and to use these vectors to predict how likely a word is given its context.', '.', '.', ', xm ).', '(1992) J. Biol.', '267, 968-974; Zhou et al.', '(1995) Mol.', '9, 208-218).', '41, 115-123) largely failed to stain tubular basement membranes, suggesting the presence of heparan sulfate chains lacking the specific JM-403 epitope.']\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import nltk\n",
|
||||||
|
"from nltk.tokenize import sent_tokenize\n",
|
||||||
|
"from nltk.tokenize import word_tokenize\n",
|
||||||
|
"f=open('sentence_splitter_input.txt','r',encoding='utf-8')\n",
|
||||||
|
"ssi=f.read()\n",
|
||||||
|
"split_result=sent_tokenize(ssi)\n",
|
||||||
|
"under=[]\n",
|
||||||
|
"over=[]\n",
|
||||||
|
"for i in split_result:\n",
|
||||||
|
" if i[0][0].istitle()==False:\n",
|
||||||
|
" print(i)\n",
|
||||||
|
" over.append(i)\n",
|
||||||
|
"print(over)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"# Q3"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 106,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"[((' a ', ' aback ', ' abandon '), 1), ((' aback ', ' abandon ', ' abandoned '), 1), ((' abandon ', ' abandoned ', ' abandoning '), 1), ((' abandoned ', ' abandoning ', ' abandonment '), 1), ((' abandoning ', ' abandonment ', ' abaringe '), 1), ((' abandonment ', ' abaringe ', ' abasement '), 1), ((' abaringe ', ' abasement ', ' abated '), 1), ((' abasement ', ' abated ', ' abatuno '), 1), ((' abated ', ' abatuno ', ' abbas '), 1), ((' abatuno ', ' abbas ', \" abbas's \"), 1)]\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import pandas as pd\n",
|
||||||
|
"from nltk import FreqDist\n",
|
||||||
|
"from nltk import ngrams\n",
|
||||||
|
"data=pd.read_csv('training set.txt',sep='\\t',header=None)\n",
|
||||||
|
"bigrams = ngrams(data[0], 3)\n",
|
||||||
|
"bigramsDist = FreqDist(bigrams)\n",
|
||||||
|
"print(bigramsDist.most_common(10))"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 108,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"[(('and', 'and', 'and'), 28933), (('a', 'a', 'a'), 23261), (('an', 'an', 'an'), 3739), (('all', 'all', 'all'), 3092), (('about', 'about', 'about'), 1815), (('after', 'after', 'after'), 1075), (('also', 'also', 'also'), 1067), (('af', 'af', 'af'), 1003), (('against', 'against', 'against'), 626), (('american', 'american', 'american'), 599)]\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"bigrams = ngrams(tri, 3)\n",
|
||||||
|
"bigramsDist = FreqDist(bigrams)\n",
|
||||||
|
"print(bigramsDist.most_common(10))"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 109,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"for key,value in bigramsDist.items():\n",
|
||||||
|
" for k in key:\n",
|
||||||
|
" if(k=='hello'):\n",
|
||||||
|
" print(key,value)\n",
|
||||||
|
" break"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 110,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"[(('a', 'aback', 'abandon'), 1), (('aback', 'abandon', 'abandoned'), 1), (('abandon', 'abandoned', 'abandoning'), 1), (('abandoned', 'abandoning', 'abandonment'), 1), (('abandoning', 'abandonment', 'abaringe'), 1), (('abandonment', 'abaringe', 'abasement'), 1), (('abaringe', 'abasement', 'abated'), 1), (('abasement', 'abated', 'abatuno'), 1), (('abated', 'abatuno', 'abbas'), 1), (('abatuno', 'abbas', \"abbas's\"), 1)]\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"from nltk import FreqDist\n",
|
||||||
|
"from nltk import ngrams\n",
|
||||||
|
"from nltk.book import text6\n",
|
||||||
|
"\n",
|
||||||
|
"bigrams = ngrams(data[0], 3)\n",
|
||||||
|
"bigramsDist = FreqDist(bigrams)\n",
|
||||||
|
"print(bigramsDist.most_common(10))\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 111,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"for key,value in bigramsDist.items():\n",
|
||||||
|
" for k in key:\n",
|
||||||
|
" if(k==' absence '):\n",
|
||||||
|
" print(key,value)\n",
|
||||||
|
" break"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## Q4"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 3,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"['Arizona.docx', 'California.docx', 'Florida.docx', 'Idaho.docx', 'Maryland.docx', 'Nevada.docx', 'New_Jersey.docx', 'North_Carolina.docx', 'Pennsylvania.docx', 'Vermont.docx', '~$rizona.docx']\n",
|
||||||
|
"=============Arizona.docx==============\n",
|
||||||
|
"['the copper state']\n",
|
||||||
|
"=============California.docx==============\n",
|
||||||
|
"['caltrans']\n",
|
||||||
|
"=============Florida.docx==============\n",
|
||||||
|
"['sunshine state']\n",
|
||||||
|
"=============Idaho.docx==============\n",
|
||||||
|
"=============Maryland.docx==============\n",
|
||||||
|
"[]\n",
|
||||||
|
"['bay state']\n",
|
||||||
|
"['bay state']\n",
|
||||||
|
"=============Nevada.docx==============\n",
|
||||||
|
"=============New_Jersey.docx==============\n",
|
||||||
|
"[]\n",
|
||||||
|
"=============North_Carolina.docx==============\n",
|
||||||
|
"=============Pennsylvania.docx==============\n",
|
||||||
|
"[]\n",
|
||||||
|
"[]\n",
|
||||||
|
"[]\n",
|
||||||
|
"=============Vermont.docx==============\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ename": "PackageNotFoundError",
|
||||||
|
"evalue": "Package not found at './state/~$rizona.docx'",
|
||||||
|
"output_type": "error",
|
||||||
|
"traceback": [
|
||||||
|
"\u001b[1;31m---------------------------------------------------------------------------\u001b[0m",
|
||||||
|
"\u001b[1;31mPackageNotFoundError\u001b[0m Traceback (most recent call last)",
|
||||||
|
"\u001b[1;32m<ipython-input-3-32ea4b732b7b>\u001b[0m in \u001b[0;36m<module>\u001b[1;34m\u001b[0m\n\u001b[0;32m 10\u001b[0m \u001b[0mgsp\u001b[0m\u001b[1;33m=\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0;32m 11\u001b[0m \u001b[1;32mfor\u001b[0m \u001b[0mstat\u001b[0m \u001b[1;32min\u001b[0m \u001b[0mstats\u001b[0m\u001b[1;33m:\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[1;32m---> 12\u001b[1;33m \u001b[0mfile\u001b[0m\u001b[1;33m=\u001b[0m\u001b[0mdocx\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0mDocument\u001b[0m\u001b[1;33m(\u001b[0m\u001b[0mdir_path\u001b[0m\u001b[1;33m+\u001b[0m\u001b[0mstat\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0m\u001b[0;32m 13\u001b[0m \u001b[0mstat_name\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0mappend\u001b[0m\u001b[1;33m(\u001b[0m\u001b[0mstat\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0msplit\u001b[0m\u001b[1;33m(\u001b[0m\u001b[1;34m'.'\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m[\u001b[0m\u001b[1;36m0\u001b[0m\u001b[1;33m]\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0;32m 14\u001b[0m \u001b[0mprint\u001b[0m\u001b[1;33m(\u001b[0m\u001b[1;34m\"=============\"\u001b[0m\u001b[1;33m+\u001b[0m\u001b[0mstat\u001b[0m\u001b[1;33m+\u001b[0m\u001b[1;34m\"==============\"\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n",
|
||||||
|
"\u001b[1;32mE:\\anaconda3\\lib\\site-packages\\docx\\api.py\u001b[0m in \u001b[0;36mDocument\u001b[1;34m(docx)\u001b[0m\n\u001b[0;32m 23\u001b[0m \"\"\"\n\u001b[0;32m 24\u001b[0m \u001b[0mdocx\u001b[0m \u001b[1;33m=\u001b[0m \u001b[0m_default_docx_path\u001b[0m\u001b[1;33m(\u001b[0m\u001b[1;33m)\u001b[0m \u001b[1;32mif\u001b[0m \u001b[0mdocx\u001b[0m \u001b[1;32mis\u001b[0m \u001b[1;32mNone\u001b[0m \u001b[1;32melse\u001b[0m \u001b[0mdocx\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[1;32m---> 25\u001b[1;33m \u001b[0mdocument_part\u001b[0m \u001b[1;33m=\u001b[0m \u001b[0mPackage\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0mopen\u001b[0m\u001b[1;33m(\u001b[0m\u001b[0mdocx\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0mmain_document_part\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0m\u001b[0;32m 26\u001b[0m \u001b[1;32mif\u001b[0m \u001b[0mdocument_part\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0mcontent_type\u001b[0m \u001b[1;33m!=\u001b[0m \u001b[0mCT\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0mWML_DOCUMENT_MAIN\u001b[0m\u001b[1;33m:\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0;32m 27\u001b[0m \u001b[0mtmpl\u001b[0m \u001b[1;33m=\u001b[0m \u001b[1;34m\"file '%s' is not a Word file, content type is '%s'\"\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n",
|
||||||
|
"\u001b[1;32mE:\\anaconda3\\lib\\site-packages\\docx\\opc\\package.py\u001b[0m in \u001b[0;36mopen\u001b[1;34m(cls, pkg_file)\u001b[0m\n\u001b[0;32m 126\u001b[0m \u001b[1;33m*\u001b[0m\u001b[0mpkg_file\u001b[0m\u001b[1;33m*\u001b[0m\u001b[1;33m.\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0;32m 127\u001b[0m \"\"\"\n\u001b[1;32m--> 128\u001b[1;33m \u001b[0mpkg_reader\u001b[0m \u001b[1;33m=\u001b[0m \u001b[0mPackageReader\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0mfrom_file\u001b[0m\u001b[1;33m(\u001b[0m\u001b[0mpkg_file\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0m\u001b[0;32m 129\u001b[0m \u001b[0mpackage\u001b[0m \u001b[1;33m=\u001b[0m \u001b[0mcls\u001b[0m\u001b[1;33m(\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0;32m 130\u001b[0m \u001b[0mUnmarshaller\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0munmarshal\u001b[0m\u001b[1;33m(\u001b[0m\u001b[0mpkg_reader\u001b[0m\u001b[1;33m,\u001b[0m \u001b[0mpackage\u001b[0m\u001b[1;33m,\u001b[0m \u001b[0mPartFactory\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n",
|
||||||
|
"\u001b[1;32mE:\\anaconda3\\lib\\site-packages\\docx\\opc\\pkgreader.py\u001b[0m in \u001b[0;36mfrom_file\u001b[1;34m(pkg_file)\u001b[0m\n\u001b[0;32m 30\u001b[0m \u001b[0mReturn\u001b[0m \u001b[0ma\u001b[0m \u001b[1;33m|\u001b[0m\u001b[0mPackageReader\u001b[0m\u001b[1;33m|\u001b[0m \u001b[0minstance\u001b[0m \u001b[0mloaded\u001b[0m \u001b[1;32mwith\u001b[0m \u001b[0mcontents\u001b[0m \u001b[0mof\u001b[0m \u001b[1;33m*\u001b[0m\u001b[0mpkg_file\u001b[0m\u001b[1;33m*\u001b[0m\u001b[1;33m.\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0;32m 31\u001b[0m \"\"\"\n\u001b[1;32m---> 32\u001b[1;33m \u001b[0mphys_reader\u001b[0m \u001b[1;33m=\u001b[0m \u001b[0mPhysPkgReader\u001b[0m\u001b[1;33m(\u001b[0m\u001b[0mpkg_file\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0m\u001b[0;32m 33\u001b[0m \u001b[0mcontent_types\u001b[0m \u001b[1;33m=\u001b[0m \u001b[0m_ContentTypeMap\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0mfrom_xml\u001b[0m\u001b[1;33m(\u001b[0m\u001b[0mphys_reader\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0mcontent_types_xml\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0;32m 34\u001b[0m \u001b[0mpkg_srels\u001b[0m \u001b[1;33m=\u001b[0m \u001b[0mPackageReader\u001b[0m\u001b[1;33m.\u001b[0m\u001b[0m_srels_for\u001b[0m\u001b[1;33m(\u001b[0m\u001b[0mphys_reader\u001b[0m\u001b[1;33m,\u001b[0m \u001b[0mPACKAGE_URI\u001b[0m\u001b[1;33m)\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n",
|
||||||
|
"\u001b[1;32mE:\\anaconda3\\lib\\site-packages\\docx\\opc\\phys_pkg.py\u001b[0m in \u001b[0;36m__new__\u001b[1;34m(cls, pkg_file)\u001b[0m\n\u001b[0;32m 29\u001b[0m \u001b[1;32melse\u001b[0m\u001b[1;33m:\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0;32m 30\u001b[0m raise PackageNotFoundError(\n\u001b[1;32m---> 31\u001b[1;33m \u001b[1;34m\"Package not found at '%s'\"\u001b[0m \u001b[1;33m%\u001b[0m \u001b[0mpkg_file\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n\u001b[0m\u001b[0;32m 32\u001b[0m )\n\u001b[0;32m 33\u001b[0m \u001b[1;32melse\u001b[0m\u001b[1;33m:\u001b[0m \u001b[1;31m# assume it's a stream and pass it to Zip reader to sort out\u001b[0m\u001b[1;33m\u001b[0m\u001b[1;33m\u001b[0m\u001b[0m\n",
|
||||||
|
"\u001b[1;31mPackageNotFoundError\u001b[0m: Package not found at './state/~$rizona.docx'"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import docx\n",
|
||||||
|
"import re\n",
|
||||||
|
"import os\n",
|
||||||
|
"#获取文档对象\n",
|
||||||
|
"dir_path=\"./state/\"\n",
|
||||||
|
"stats=os.listdir(dir_path)\n",
|
||||||
|
"print(stats)\n",
|
||||||
|
"stat_name=[]\n",
|
||||||
|
"nickname=[]\n",
|
||||||
|
"gsp=[]\n",
|
||||||
|
"for stat in stats:\n",
|
||||||
|
" file=docx.Document(dir_path+stat)\n",
|
||||||
|
" stat_name.append(stat.split('.')[0])\n",
|
||||||
|
" print(\"=============\"+stat+\"==============\")\n",
|
||||||
|
" n_name=[]\n",
|
||||||
|
" n_money=[]\n",
|
||||||
|
" for para in file.paragraphs:\n",
|
||||||
|
" para=para.text.lower()\n",
|
||||||
|
" # print(re.findall(r'nickname(.*?)', para))\n",
|
||||||
|
" \n",
|
||||||
|
" para_t=sent_tokenize(para)\n",
|
||||||
|
" for p in para_t:\n",
|
||||||
|
" if 'gross state product' in p:\n",
|
||||||
|
" money=re.findall('\\$\\d+\\.\\d+',p,re.S)\n",
|
||||||
|
" n_money.extend(money)\n",
|
||||||
|
" if 'nickname' in p:\n",
|
||||||
|
" n_name.extend(re.findall('nickname.*?\\\"(.*?)\"',p,re.S))\n",
|
||||||
|
" print(n_name)\n",
|
||||||
|
" nickname.append(n_name)\n",
|
||||||
|
" gsp.append(n_money)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 190,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [],
|
||||||
|
"source": [
|
||||||
|
"# print(gsp)\n",
|
||||||
|
"# print(nickname)\n",
|
||||||
|
"# print(stat_name)\n",
|
||||||
|
"for i,j,k in zip(gsp,nickname,stat_name):\n",
|
||||||
|
" if i==None:\n",
|
||||||
|
" print(i)\n",
|
||||||
|
" print(6)\n",
|
||||||
|
"res=pd.DataFrame({\n",
|
||||||
|
" 'stat':stat_name,\n",
|
||||||
|
" 'nickname':nickname,\n",
|
||||||
|
" 'gsp':gsp\n",
|
||||||
|
"})\n",
|
||||||
|
"res.to_csv('res.tsv',sep='\\t',index=None)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 13,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"=============~$rizona.docx==============\n",
|
||||||
|
"['the copper state']\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"file=docx.Document(dir_path+'Arizona.docx')\n",
|
||||||
|
"stat_name.append(stat.split('.')[0])\n",
|
||||||
|
"print(\"=============\"+stat+\"==============\")\n",
|
||||||
|
"n_name=[]\n",
|
||||||
|
"n_money=[]\n",
|
||||||
|
"for para in file.paragraphs:\n",
|
||||||
|
" para=para.text.lower()\n",
|
||||||
|
"# print(re.findall(r'nickname(.*?)', para))\n",
|
||||||
|
"\n",
|
||||||
|
" para_t=sent_tokenize(para)\n",
|
||||||
|
" for p in para_t:\n",
|
||||||
|
" if 'gross state product' in p:\n",
|
||||||
|
" money=re.findall('\\$\\d+\\.\\d+',p,re.S)\n",
|
||||||
|
" n_money.extend(money)\n",
|
||||||
|
" if 'nickname' in p:\n",
|
||||||
|
" n_name.extend(re.findall('nickname.*?\\\"(.*?)\"',p,re.S))\n",
|
||||||
|
" print(n_name)\n",
|
||||||
|
"# tables=file.tables\n",
|
||||||
|
"# for i in range(len(tables)):\n",
|
||||||
|
"# tb=tables[i]\n",
|
||||||
|
"# #获取表格的行\n",
|
||||||
|
"# tb_rows=tb.rows\n",
|
||||||
|
"# #读取每一行内容\n",
|
||||||
|
"# for i in range(len(tb_rows)):\n",
|
||||||
|
"# row_data=[]\n",
|
||||||
|
"# row_cells=tb_rows[i].cells\n",
|
||||||
|
"# #读取每一行单元格内容\n",
|
||||||
|
"# for cell in row_cells:\n",
|
||||||
|
"# #单元格内容\n",
|
||||||
|
"# row_data.append(cell.text)\n",
|
||||||
|
"# print(row_data)\n",
|
||||||
|
"children = file.element.body.iter()\n",
|
||||||
|
"child_iters = []\n",
|
||||||
|
"for child in children:\n",
|
||||||
|
" # 通过类型判断目录\n",
|
||||||
|
" if child.tag.endswith('textbox'):\n",
|
||||||
|
" for ci in child.iter():\n",
|
||||||
|
" if ci.tag.endswith('main}r'):\n",
|
||||||
|
" child_iters.append(ci)\n",
|
||||||
|
"textbox = [ci.text for ci in child_iters]\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 18,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"Arizona\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": []
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"metadata": {
|
||||||
|
"kernelspec": {
|
||||||
|
"display_name": "Python 3",
|
||||||
|
"language": "python",
|
||||||
|
"name": "python3"
|
||||||
|
},
|
||||||
|
"language_info": {
|
||||||
|
"codemirror_mode": {
|
||||||
|
"name": "ipython",
|
||||||
|
"version": 3
|
||||||
|
},
|
||||||
|
"file_extension": ".py",
|
||||||
|
"mimetype": "text/x-python",
|
||||||
|
"name": "python",
|
||||||
|
"nbconvert_exporter": "python",
|
||||||
|
"pygments_lexer": "ipython3",
|
||||||
|
"version": "3.7.1"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"nbformat": 4,
|
||||||
|
"nbformat_minor": 2
|
||||||
|
}
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
,stat,nickname,gsp
|
||||||
|
0,Arizona,['the copper state'],[]
|
||||||
|
1,California,['caltrans'],[]
|
||||||
|
2,Florida,['sunshine state'],[]
|
||||||
|
3,Idaho,[],[]
|
||||||
|
4,Maryland,['bay state'],['$382.4']
|
||||||
|
5,Nevada,[],[]
|
||||||
|
6,New_Jersey,[],[]
|
||||||
|
7,North_Carolina,[],"['$424.9', '$2.4', '$57.8']"
|
||||||
|
8,Pennsylvania,[],[]
|
||||||
|
9,Vermont,[],[]
|
||||||
|
Can't render this file because it contains an unexpected character in line 9 and column 21.
|
@@ -0,0 +1,13 @@
|
|||||||
|
Prior to joining the Navy, Hopper earned a Ph.D. in mathematics from Yale University and was a professor of mathematics at Vassar College. Hopper attempted to enlist in the Navy during World War II but was rejected because she was 34 years old. She instead joined the Navy Reserves. Hopper began her computing career in 1944 when she worked on the Harvard Mark I team led by Howard H. Aiken. In 1949, she joined the Eckert–Mauchly Computer Corporation and was part of the team that developed the UNIVAC I computer. At Eckert–Mauchly she began developing the compiler. She believed that a programming language based on English was possible. Her compiler converted English terms into machine code understood by computers.By 1952, Hopper had finished her program linker (originally called a compiler), which was written for the A-0 System. During her wartime service, she co-authored three papers based on her work on the Harvard Mark 1. In 1954, Eckert–Mauchly chose Hopper to lead their department for automatic programming, and she led the release of some of the first compiled languages like FLOW-MATIC. In 1959, she participated in the CODASYL consortium, which consulted Hopper to guide them in creating a machine-independent programming language. This led to the COBOL language, which was inspired by her idea of a language being based on English words.In 1966, she retired from the Naval Reserve, but in 1967 the Navy recalled her to active duty.
|
||||||
|
|
||||||
|
A carbohydrate structural variant of MM glycoprotein (glycophorin A). A variant of the MM glycoprotein (glycophorin A) was isolated from erythrocyte membranes of two individual donors, a mother (L.G.) and daughter (V.W.). This glycoprotein was found to be a carbohydrate variant in which, for both donors, certain O-glycosidically linked saccharides retained the core structure consisting of NeuAc(alpha 2,3)Gal(beta 1,3)GalNAc that is common to all O-linked saccharides of the MN glycoproteins, and, in addition, contained substituents, of varying chain lengths, on the primary carbinol of GalNAc. These saccharides were released from the polypeptide by beta-elimination in the presence of sodium borohydride, and aspects of their structure were investigated by glycosidase digestion and periodate oxidation. Thus, the smallest variant structure was deduced to be NeuAc(alpha 2,3)Gal(beta 1,3)[GlcNAc(beta 1,6)]H2GalNAc. The 6-O-linked GlcNAc appears to serve as the focus of further chain elongation reactions, involving alternate additions of Gal and GlcNAc residues and leading to the formation of several homologous structures. Two such structures, NeuAc(alpha 2,3)Gal(beta 1,3)[GlcNAc(beta 1,?) Gal(beta 1,3/4)GlcNAc(beta 1,6)]H2GalNAc and NeuAc(alpha 2,3) Gal(beta 1,3)[Gal(beta 1,3/4)GlcNAc(beta 1,6)]H2GalNAc were the predominant species present. A larger saccharide was also isolated and its partial sequence was determined to be Gal(beta 1,3/4)GlcNAc(beta 1,?)[Gal(beta 1,3/4)Glc-NAc(beta 1,?)] Gal(beta 1,3/4)GlcNAc(beta 1,6)[NeuAc(alpha 2,3)Gal-(beta 1,3)]H2GalNAc. Because the peptide portion of these glycoproteins contains two methionine residues, it was possible to isolate two CNBr glycopeptides from separate regions of the molecule, and to assess the distribution of these variant structures in the polypeptide.
|
||||||
|
|
||||||
|
Characterization and Antioxidant Activity Determination of Neutral and Acidic Polysaccharides from Panax Ginseng C. A. Meyer. Panax ginseng (P. ginseng) is the most widely consumed herbal plant in Asia and is well-known for its various pharmacological properties. Many studies have been devoted to this natural product. However, polysaccharide's components of ginseng and their biological effects have not been widely studied. In this study, white ginseng neutral polysaccharide (WGNP) and white ginseng acidic polysaccharide (WGAP) fractions were purified from P. ginseng roots. The chemical properties of WGNP and WGAP were investigated using various chromatography and spectroscopy techniques, including high-performance gel permeation chromatography, Fourier-transform infrared spectroscopy, and high-performance liquid chromatography with an ultra-violet detector. The antioxidant, anti-radical, and hydrogen peroxide scavenging activities were evaluated in vitro and in vivo using Caenorhabditis elegans as the model organism. Our in vitro data by ABTS (2,2'-azino-bis-(3-ethylbenzothiazoline-6-sulfonic acid), reducing power, ferrous ion chelating, and hydroxyl radical scavenging activity suggested that the WGAP with significantly higher uronic acid content and higher molecular weight exhibits a much stronger antioxidant effect as compared to that of WGNP. Similar antioxidant activity of WGAP was also confirmed in vivo by evaluating internal reactive oxygen species (ROS) concentration and lipid peroxidation. In conclusion, WGAP may be used as a natural antioxidant with potent scavenging and metal chelation properties.
|
||||||
|
|
||||||
|
Protein Thermodynamic Destabilization in the Assessment of Pathogenicity of a Variant of Uncertain Significance in Cardiac Myosin Binding Protein C. In the era of next generation sequencing (NGS), genetic testing for inherited disorders identifies an ever-increasing number of variants whose pathogenicity remains unclear. These variants of uncertain significance (VUS) limit the reach of genetic testing in clinical practice. The VUS for hypertrophic cardiomyopathy (HCM), the most common familial heart disease, constitute over 60% of entries for missense variants shown in ClinVar database. We have studied a novel VUS (c.1809T>G-p.I603M) in the most frequently mutated gene in HCM, MYBPC3, which codes for cardiac myosin-binding protein C (cMyBPC). Our determinations of pathogenicity integrate bioinformatics evaluation and functional studies of RNA splicing and protein thermodynamic stability. In silico prediction and mRNA analysis indicated no alteration of RNA splicing induced by the variant. At the protein level, the p.I603M mutation maps to the C4 domain of cMyBPC. Although the mutation does not perturb much the overall structure of the C4 domain, the stability of C4 I603M is severely compromised as detected by circular dichroism and differential scanning calorimetry experiments. Taking into account the highly destabilizing effect of the mutation in the structure of C4, we propose reclassification of variant p.I603M as likely pathogenic. Looking into the future, the workflow described here can be used to refine the assignment of pathogenicity of variants of uncertain significance in MYBPC3.
|
||||||
|
|
||||||
|
The idea of neural language models as introduced by Bengio et al. [5] is to jointly learn an em- bedding of words into an n-dimensional vector space and to use these vectors to predict how likely a word is given its context. Collobert and Weston [6] introduced a new neural network model to compute such an embedding. When these networks are optimized via gradient ascent the derivatives modify the word embedding matrix L ∈ Rn×|V |, where |V | is the size of the vocabulary. The word vectors inside the embedding matrix capture distributional syntactic and semantic information via the word’s co-occurrence statistics. For further details and evaluations of these embeddings, see [5, 6, 7, 8]. Once this matrix is learned on an unlabeled corpus, we can use it for subsequent tasks by using each word’s vector (a column in L) to represent that word. In the remainder of this paper, we represent a sentence (or any n-gram) as an ordered list of these vectors (x1 , . . . , xm ). This word representation is better suited for autoencoders than the binary number representations used in previous related autoencoder models such as the recursive autoassociative memory (RAAM) model of Pollack [9, 10] or recurrent neural networks [11] since the activations are inherently continuous.
|
||||||
|
|
||||||
|
The sequences mediating receptor mRNA down-regulation are represented within the AR cDNA and not within the CMV promoter. Androgenic down-regulation of AR cDNA expression was time- and dose-dependent, resembling native AR mRNA down-regulation. In addition, androgenic regulation of the receptor cDNA was not dependent on protein synthesis suggesting that AR and/or another pre-existing protein(s) is involved in this process. In COS 1 cells co-transfected with androgen and glucocorticoid receptor cDNAs, dexamethasone mimicked the action of androgen in down-regulating AR mRNA. This response depended on glucocorticoid receptors. Androgen had little effect on steady-state levels of AR protein consistent with reports that androgen down-regulates AR mRNA but increases AR protein half-life (Kemppainen et al. (1992) J. Biol. Chem. 267, 968-974; Zhou et al. (1995) Mol. Endocrinol. 9, 208-218). However, glucocorticoids decreased AR protein levels in cells that co-expressed androgen and glucocorticoid receptors. These results indicate that sequences represented in the AR cDNA mediate AR mRNA down-regulation by both androgens and glucocorticoids. Inhibition of AR mRNA and protein by glucocorticoids suggests that these steroids may modulate androgen action in tissues, such as mammary gland and prostate, which express both androgen and glucocorticoid receptors.
|
||||||
|
|
||||||
|
Immunohistochemical application of antibodies against heparan sulfate proteoglycan core protein and heparitinase-digested heparan sulfate stubs showed the presence of heparan sulfate proteoglycan in all basement membranes of the rat kidney. However, a monoclonal antibody (JM-403) against native heparan sulfate (van den Born, J., van den Heuvel, L. P. W. J., Bakker, M. A. H., Veerkamp, J. H., Assmann, K. J. M., and Berden, J. H. M. (1992) Kidney Int. 41, 115-123) largely failed to stain tubular basement membranes, suggesting the presence of heparan sulfate chains lacking the specific JM-403 epitope. Heparan sulfate preparations from various sources differed markedly with regard to JM-403 binding, as demonstrated by liquid phase inhibition in enzyme-linked immunosorbent assay, the interaction decreasing with increasing sulfate contents of the polysaccharide. Mapping of the JM-403 epitope indicated that it was dominated by one or more N-unsubstituted glucosamine unit(s), since treatments that destroyed or altered the structure of such units in heparan sulfate preparations (cleavage at N-unsubstituted glucosamine units with HNO2 at pH 3.9 and N-acetylation with acetic anhydride, respectively), abolished antibody binding.
|
||||||
|
After Width: | Height: | Size: 179 KiB |
@@ -0,0 +1,485 @@
|
|||||||
|
{
|
||||||
|
"cells": [
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## Q2"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 1,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"and daughter (V.W.).\n",
|
||||||
|
")[Gal(beta 1,3/4)Glc-NAc(beta 1,?)]\n",
|
||||||
|
"[5] is to jointly learn an em- bedding of words into an n-dimensional vector space and to use these vectors to predict how likely a word is given its context.\n",
|
||||||
|
".\n",
|
||||||
|
".\n",
|
||||||
|
", xm ).\n",
|
||||||
|
"(1992) J. Biol.\n",
|
||||||
|
"267, 968-974; Zhou et al.\n",
|
||||||
|
"(1995) Mol.\n",
|
||||||
|
"9, 208-218).\n",
|
||||||
|
"41, 115-123) largely failed to stain tubular basement membranes, suggesting the presence of heparan sulfate chains lacking the specific JM-403 epitope.\n",
|
||||||
|
"['and daughter (V.W.).', ')[Gal(beta 1,3/4)Glc-NAc(beta 1,?)]', '[5] is to jointly learn an em- bedding of words into an n-dimensional vector space and to use these vectors to predict how likely a word is given its context.', '.', '.', ', xm ).', '(1992) J. Biol.', '267, 968-974; Zhou et al.', '(1995) Mol.', '9, 208-218).', '41, 115-123) largely failed to stain tubular basement membranes, suggesting the presence of heparan sulfate chains lacking the specific JM-403 epitope.']\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import nltk\n",
|
||||||
|
"from nltk.tokenize import sent_tokenize\n",
|
||||||
|
"from nltk.tokenize import word_tokenize\n",
|
||||||
|
"f=open('sentence_splitter_input.txt','r',encoding='utf-8')\n",
|
||||||
|
"ssi=f.read()\n",
|
||||||
|
"split_result=sent_tokenize(ssi)\n",
|
||||||
|
"under=[]\n",
|
||||||
|
"over=[]\n",
|
||||||
|
"for i in split_result:\n",
|
||||||
|
" if i[0][0].istitle()==False:\n",
|
||||||
|
" print(i)\n",
|
||||||
|
" over.append(i)\n",
|
||||||
|
"print(over)"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"# Q3"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 96,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stderr",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"E:\\anaconda3\\lib\\site-packages\\ipykernel_launcher.py:49: SettingWithCopyWarning: \n",
|
||||||
|
"A value is trying to be set on a copy of a slice from a DataFrame\n",
|
||||||
|
"\n",
|
||||||
|
"See the caveats in the documentation: http://pandas.pydata.org/pandas-docs/stable/user_guide/indexing.html#returning-a-view-versus-a-copy\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
"===============================hello====================================\n",
|
||||||
|
"e is the three most likely characters after the bigram h and the rank of the actual observed character e is 5.26%\n",
|
||||||
|
"t is the three most likely characters after the bigram e and the rank of the actual observed character l is 1.09%\n",
|
||||||
|
"e is the three most likely characters after the bigram l and the rank of the actual observed character l is 1.04%\n",
|
||||||
|
"e is the three most likely characters after the bigram l and the rank of the actual observed character o is 0.64%\n",
|
||||||
|
"f is the three most likely characters after the bigram o and the rank of the actual observed character is 0.64%\n",
|
||||||
|
"e is the three most likely characters after the bigram he and the rank of the actual observed character e is 0.03%\n",
|
||||||
|
"i is the three most likely characters after the bigram el and the rank of the actual observed character l is 0.08%\n",
|
||||||
|
"l is the three most likely characters after the bigram ll and the rank of the actual observed character l is 0.04%\n",
|
||||||
|
"===============================weather====================================\n",
|
||||||
|
"a is the three most likely characters after the bigram w and the rank of the actual observed character e is 0.60%\n",
|
||||||
|
"t is the three most likely characters after the bigram e and the rank of the actual observed character a is 1.54%\n",
|
||||||
|
"n is the three most likely characters after the bigram a and the rank of the actual observed character t is 2.25%\n",
|
||||||
|
"h is the three most likely characters after the bigram t and the rank of the actual observed character h is 5.70%\n",
|
||||||
|
"e is the three most likely characters after the bigram h and the rank of the actual observed character e is 5.26%\n",
|
||||||
|
"t is the three most likely characters after the bigram e and the rank of the actual observed character r is 3.35%\n",
|
||||||
|
"e is the three most likely characters after the bigram r and the rank of the actual observed character is 3.35%\n",
|
||||||
|
"e is the three most likely characters after the bigram we and the rank of the actual observed character e is 0.01%\n",
|
||||||
|
"a is the three most likely characters after the bigram ea and the rank of the actual observed character a is 0.09%\n",
|
||||||
|
"t is the three most likely characters after the bigram at and the rank of the actual observed character t is 0.05%\n",
|
||||||
|
"e is the three most likely characters after the bigram th and the rank of the actual observed character h is 1.95%\n",
|
||||||
|
"e is the three most likely characters after the bigram he and the rank of the actual observed character e is 0.33%\n",
|
||||||
|
"===============================mystique====================================\n",
|
||||||
|
"e is the three most likely characters after the bigram m and the rank of the actual observed character y is 0.09%\n",
|
||||||
|
"b is the three most likely characters after the bigram y and the rank of the actual observed character s is 0.36%\n",
|
||||||
|
"i is the three most likely characters after the bigram s and the rank of the actual observed character t is 2.18%\n",
|
||||||
|
"h is the three most likely characters after the bigram t and the rank of the actual observed character i is 2.27%\n",
|
||||||
|
"n is the three most likely characters after the bigram i and the rank of the actual observed character q is 0.02%\n",
|
||||||
|
"u is the three most likely characters after the bigram q and the rank of the actual observed character u is 0.21%\n",
|
||||||
|
"l is the three most likely characters after the bigram u and the rank of the actual observed character e is 0.21%\n",
|
||||||
|
"t is the three most likely characters after the bigram e and the rank of the actual observed character is 0.21%\n",
|
||||||
|
"y is the three most likely characters after the bigram my and the rank of the actual observed character y is 0.00%\n",
|
||||||
|
"s is the three most likely characters after the bigram ys and the rank of the actual observed character s is 0.04%\n",
|
||||||
|
"t is the three most likely characters after the bigram st and the rank of the actual observed character t is 0.12%\n",
|
||||||
|
"o is the three most likely characters after the bigram ti and the rank of the actual observed character i is 0.00%\n",
|
||||||
|
"u is the three most likely characters after the bigram iq and the rank of the actual observed character q is 0.01%\n",
|
||||||
|
"u is the three most likely characters after the bigram qu and the rank of the actual observed character u is 0.04%\n",
|
||||||
|
"===============================catcher====================================\n",
|
||||||
|
"o is the three most likely characters after the bigram c and the rank of the actual observed character a is 0.83%\n",
|
||||||
|
"n is the three most likely characters after the bigram a and the rank of the actual observed character t is 2.25%\n",
|
||||||
|
"h is the three most likely characters after the bigram t and the rank of the actual observed character c is 0.23%\n",
|
||||||
|
"o is the three most likely characters after the bigram c and the rank of the actual observed character h is 0.92%\n",
|
||||||
|
"e is the three most likely characters after the bigram h and the rank of the actual observed character e is 5.26%\n",
|
||||||
|
"t is the three most likely characters after the bigram e and the rank of the actual observed character r is 3.35%\n",
|
||||||
|
"e is the three most likely characters after the bigram r and the rank of the actual observed character is 3.35%\n",
|
||||||
|
"a is the three most likely characters after the bigram ca and the rank of the actual observed character a is 0.06%\n",
|
||||||
|
"t is the three most likely characters after the bigram at and the rank of the actual observed character t is 0.02%\n",
|
||||||
|
"o is the three most likely characters after the bigram tc and the rank of the actual observed character c is 0.03%\n",
|
||||||
|
"h is the three most likely characters after the bigram ch and the rank of the actual observed character h is 0.08%\n",
|
||||||
|
"e is the three most likely characters after the bigram he and the rank of the actual observed character e is 0.33%\n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import pandas as pd\n",
|
||||||
|
"from nltk import FreqDist\n",
|
||||||
|
"from nltk import ngrams\n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
"def to_df(FQ):\n",
|
||||||
|
" '''\n",
|
||||||
|
" Translate FreqDist to DataFrame\n",
|
||||||
|
" '''\n",
|
||||||
|
" key_word=[]\n",
|
||||||
|
" frequence=[]\n",
|
||||||
|
" for key,value in FQ.items():\n",
|
||||||
|
" key=''.join(key)\n",
|
||||||
|
" key_word.append(key)\n",
|
||||||
|
" frequence.append(value)\n",
|
||||||
|
"\n",
|
||||||
|
" data=pd.DataFrame({\n",
|
||||||
|
" 'key_words':key_word,\n",
|
||||||
|
" 'frequence':frequence\n",
|
||||||
|
" })\n",
|
||||||
|
" return data\n",
|
||||||
|
"\n",
|
||||||
|
"def get_word1(key_word):\n",
|
||||||
|
" all_count=freq['frequence'].sum()\n",
|
||||||
|
" if len(key_word)==1:\n",
|
||||||
|
" next_c=freq[freq['key_words'].str[0]==key_word]['key_words'].tolist()[0][1]\n",
|
||||||
|
" new_key=key_word+next_c\n",
|
||||||
|
" key_p=freq[freq['key_words'].str.contains(new_key)]['frequence'].sum()\n",
|
||||||
|
" return next_c,key_p/all_count\n",
|
||||||
|
" \n",
|
||||||
|
" elif(len(key_word)==2):\n",
|
||||||
|
" next_c=freq[freq['key_words'].str.contains(key_word)]['key_words'].tolist()[0][2]\n",
|
||||||
|
" new_key=key_word+next_c\n",
|
||||||
|
" key_p=freq[freq['key_words'].str.contains(new_key)]['frequence'].sum()\n",
|
||||||
|
" return next_c,key_p/all_count\n",
|
||||||
|
"\n",
|
||||||
|
"def get_word2(current_word,next_word):\n",
|
||||||
|
" all_count=freq['frequence'].sum()\n",
|
||||||
|
" new_key=current_word+next_word\n",
|
||||||
|
" key_p=freq[freq['key_words'].str.contains(new_key)]['frequence'].sum()\n",
|
||||||
|
" return key_p/all_count\n",
|
||||||
|
"data=pd.read_csv('training set.txt',sep='\\t',header=None)\n",
|
||||||
|
"\n",
|
||||||
|
"words=\"\"\n",
|
||||||
|
"data[2]=data[0]*data[1]\n",
|
||||||
|
"# words+=''.join(data[0]*data[1][i])\n",
|
||||||
|
"data[2]=data[2].str.strip()\n",
|
||||||
|
"for i in range(len(data)):\n",
|
||||||
|
" data[2][i]=word_tokenize(data[2][i])\n",
|
||||||
|
" \n",
|
||||||
|
"words=[]\n",
|
||||||
|
"for i in data[2]:\n",
|
||||||
|
" words.extend(i)\n",
|
||||||
|
"words=''.join(words)\n",
|
||||||
|
"bigrams = ngrams(words, 3)\n",
|
||||||
|
"bigramsDist = FreqDist(bigrams)\n",
|
||||||
|
"\n",
|
||||||
|
"freq=to_df(bigramsDist)\n",
|
||||||
|
"freq=freq.sort_values('frequence',ascending=False)\n",
|
||||||
|
"train_set=['hello','weather','mystique','catcher']\n",
|
||||||
|
"for t_s in train_set:\n",
|
||||||
|
" my_word=t_s\n",
|
||||||
|
" print(\"===============================\"+my_word+\"====================================\")\n",
|
||||||
|
" for i in range(len(my_word)):\n",
|
||||||
|
"\n",
|
||||||
|
" next_c,p_value=get_word1(my_word[i])\n",
|
||||||
|
" try:\n",
|
||||||
|
" p_value2=get_word2(my_word[i],my_word[i+1])\n",
|
||||||
|
" except:\n",
|
||||||
|
" print('{} is the three most likely characters after the bigram {} and the rank of the actual observed character {} is {:.2f}%'.format(next_c,my_word[i],'',p_value2*100))\n",
|
||||||
|
" continue \n",
|
||||||
|
" print('{} is the three most likely characters after the bigram {} and the rank of the actual observed character {} is {:.2f}%'.format(next_c,my_word[i],my_word[i+1],p_value2*100))\n",
|
||||||
|
" for i in range(0,len(my_word)):\n",
|
||||||
|
" words=my_word[i:i+2]\n",
|
||||||
|
" next_c,p_value=get_word1(words)\n",
|
||||||
|
" try:\n",
|
||||||
|
" p_value2=get_word2(words,my_word[i+2])\n",
|
||||||
|
" except:\n",
|
||||||
|
" continue\n",
|
||||||
|
" print('{} is the three most likely characters after the bigram {} and the rank of the actual observed character {} is {:.2f}%'.format(next_c,words,my_word[i+1],p_value2*100))\n",
|
||||||
|
" \n",
|
||||||
|
" "
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## Q4"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 4,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"name": "stdout",
|
||||||
|
"output_type": "stream",
|
||||||
|
"text": [
|
||||||
|
" stat gsp \\\n",
|
||||||
|
"0 Arizona $259 billion \n",
|
||||||
|
"1 California $3.0 trillion \n",
|
||||||
|
"2 Florida $1.0 trillion \n",
|
||||||
|
"3 Idaho None \n",
|
||||||
|
"4 Maryland $382.4 billion \n",
|
||||||
|
"5 Nevada None \n",
|
||||||
|
"6 New_Jersey None \n",
|
||||||
|
"7 North_Carolina $496 billion \n",
|
||||||
|
"8 Pennsylvania $803 billion \n",
|
||||||
|
"9 Vermont None \n",
|
||||||
|
"\n",
|
||||||
|
" nickname \n",
|
||||||
|
"0 The Grand Canyon State; The Copper State;The V... \n",
|
||||||
|
"1 The Golden State \n",
|
||||||
|
"2 The Sunshine State \n",
|
||||||
|
"3 Gem State \n",
|
||||||
|
"4 Old Line State; Free State; Little America; Am... \n",
|
||||||
|
"5 None \n",
|
||||||
|
"6 The Garden State \n",
|
||||||
|
"7 Old North State; Tar Heel State \n",
|
||||||
|
"8 Keystone State; Quaker State \n",
|
||||||
|
"9 The Green Mountain State \n"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"import docx\n",
|
||||||
|
"import re\n",
|
||||||
|
"import os\n",
|
||||||
|
"import pandas as pd\n",
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
"def nickname_para(file):\n",
|
||||||
|
" children = file.element.body.iter()\n",
|
||||||
|
" child_iters = []\n",
|
||||||
|
" for child in children:\n",
|
||||||
|
" # 通过类型判断目录\n",
|
||||||
|
" if child.tag.endswith('textbox'):\n",
|
||||||
|
" for ci in child.iter():\n",
|
||||||
|
" if ci.tag.endswith('main}r'):\n",
|
||||||
|
" child_iters.append(ci)\n",
|
||||||
|
" textbox = [ci.text for ci in child_iters]\n",
|
||||||
|
" p=''.join(textbox)\n",
|
||||||
|
" pattern=re.compile('Nickname\\(s\\): (.*?)\\(s\\)')\n",
|
||||||
|
" return re.findall(pattern,p)\n",
|
||||||
|
"def gsp(file):\n",
|
||||||
|
" for p in file.paragraphs:\n",
|
||||||
|
" xml = p.paragraph_format.element.xml\n",
|
||||||
|
" xml_str = str(xml)\n",
|
||||||
|
" wt_list = re.findall('<w:t[\\S\\s]*?</w:t>', xml_str)\n",
|
||||||
|
" hyperlink = u''\n",
|
||||||
|
" for wt in wt_list:\n",
|
||||||
|
" wt_content = re.sub('<[\\S\\s]*?>', u'', wt)\n",
|
||||||
|
" hyperlink += wt_content\n",
|
||||||
|
" gsp=re.findall('gross state product.*?(\\$\\d+\\.\\d+)',str(hyperlink))\n",
|
||||||
|
" \n",
|
||||||
|
" if 'gross state product' in hyperlink:\n",
|
||||||
|
" res=re.findall('(\\$[\\d+\\.]+ billion|\\$[\\d+\\.]+ trillion)',hyperlink)\n",
|
||||||
|
" return res\n",
|
||||||
|
"#获取文档对象\n",
|
||||||
|
"dir_path=\"./state/\"\n",
|
||||||
|
"stats=os.listdir(dir_path)\n",
|
||||||
|
"\n",
|
||||||
|
"stat_name=[]\n",
|
||||||
|
"nickname=[]\n",
|
||||||
|
"gross=[]\n",
|
||||||
|
"\n",
|
||||||
|
"for stat in stats:\n",
|
||||||
|
" try:\n",
|
||||||
|
" file=docx.Document(dir_path+stat)\n",
|
||||||
|
" except:\n",
|
||||||
|
" continue\n",
|
||||||
|
" stat_name.append(stat.split('.')[0])\n",
|
||||||
|
" nickname.append(nickname_para(file))\n",
|
||||||
|
" gross.append(gsp(file))\n",
|
||||||
|
"\n",
|
||||||
|
"res=pd.DataFrame({\n",
|
||||||
|
" 'stat':stat_name,\n",
|
||||||
|
" 'gsp':gross,\n",
|
||||||
|
" 'nickname':nickname\n",
|
||||||
|
"})\n",
|
||||||
|
"\n",
|
||||||
|
"for i in range(len(res)):\n",
|
||||||
|
" try:\n",
|
||||||
|
" res['nickname'][i]=res['nickname'][i][0]\n",
|
||||||
|
" res['nickname'][i]=res['nickname'][i].replace(\"\\\"\",\"\")\n",
|
||||||
|
" res['nickname'][i]=res['nickname'][i].replace(\",\",\";\")\n",
|
||||||
|
" res['nickname'][i]=res['nickname'][i].replace(\"Motto\",\"\")\n",
|
||||||
|
" res['nickname'][i]=res['nickname'][i].replace(\"[1]\",\"\")\n",
|
||||||
|
" res['nickname'][i]=res['nickname'][i].replace(\"[2]\",\"\")\n",
|
||||||
|
" except:\n",
|
||||||
|
" res['nickname'][i]='None'\n",
|
||||||
|
" try:\n",
|
||||||
|
" res['gsp'][i]=res['gsp'][i][0]\n",
|
||||||
|
" except:\n",
|
||||||
|
" res['gsp'][i]='None'\n",
|
||||||
|
" \n",
|
||||||
|
"res.to_csv('res.tsv',sep='\\t',index=None)\n",
|
||||||
|
"print(res) \n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"execution_count": 5,
|
||||||
|
"metadata": {},
|
||||||
|
"outputs": [
|
||||||
|
{
|
||||||
|
"data": {
|
||||||
|
"text/html": [
|
||||||
|
"<div>\n",
|
||||||
|
"<style scoped>\n",
|
||||||
|
" .dataframe tbody tr th:only-of-type {\n",
|
||||||
|
" vertical-align: middle;\n",
|
||||||
|
" }\n",
|
||||||
|
"\n",
|
||||||
|
" .dataframe tbody tr th {\n",
|
||||||
|
" vertical-align: top;\n",
|
||||||
|
" }\n",
|
||||||
|
"\n",
|
||||||
|
" .dataframe thead th {\n",
|
||||||
|
" text-align: right;\n",
|
||||||
|
" }\n",
|
||||||
|
"</style>\n",
|
||||||
|
"<table border=\"1\" class=\"dataframe\">\n",
|
||||||
|
" <thead>\n",
|
||||||
|
" <tr style=\"text-align: right;\">\n",
|
||||||
|
" <th></th>\n",
|
||||||
|
" <th>stat</th>\n",
|
||||||
|
" <th>gsp</th>\n",
|
||||||
|
" <th>nickname</th>\n",
|
||||||
|
" </tr>\n",
|
||||||
|
" </thead>\n",
|
||||||
|
" <tbody>\n",
|
||||||
|
" <tr>\n",
|
||||||
|
" <td>0</td>\n",
|
||||||
|
" <td>Arizona</td>\n",
|
||||||
|
" <td>$259 billion</td>\n",
|
||||||
|
" <td>The Grand Canyon State; The Copper State;The V...</td>\n",
|
||||||
|
" </tr>\n",
|
||||||
|
" <tr>\n",
|
||||||
|
" <td>1</td>\n",
|
||||||
|
" <td>California</td>\n",
|
||||||
|
" <td>$3.0 trillion</td>\n",
|
||||||
|
" <td>The Golden State</td>\n",
|
||||||
|
" </tr>\n",
|
||||||
|
" <tr>\n",
|
||||||
|
" <td>2</td>\n",
|
||||||
|
" <td>Florida</td>\n",
|
||||||
|
" <td>$1.0 trillion</td>\n",
|
||||||
|
" <td>The Sunshine State</td>\n",
|
||||||
|
" </tr>\n",
|
||||||
|
" <tr>\n",
|
||||||
|
" <td>3</td>\n",
|
||||||
|
" <td>Idaho</td>\n",
|
||||||
|
" <td>None</td>\n",
|
||||||
|
" <td>Gem State</td>\n",
|
||||||
|
" </tr>\n",
|
||||||
|
" <tr>\n",
|
||||||
|
" <td>4</td>\n",
|
||||||
|
" <td>Maryland</td>\n",
|
||||||
|
" <td>$382.4 billion</td>\n",
|
||||||
|
" <td>Old Line State; Free State; Little America; Am...</td>\n",
|
||||||
|
" </tr>\n",
|
||||||
|
" <tr>\n",
|
||||||
|
" <td>5</td>\n",
|
||||||
|
" <td>Nevada</td>\n",
|
||||||
|
" <td>None</td>\n",
|
||||||
|
" <td>None</td>\n",
|
||||||
|
" </tr>\n",
|
||||||
|
" <tr>\n",
|
||||||
|
" <td>6</td>\n",
|
||||||
|
" <td>New_Jersey</td>\n",
|
||||||
|
" <td>None</td>\n",
|
||||||
|
" <td>The Garden State</td>\n",
|
||||||
|
" </tr>\n",
|
||||||
|
" <tr>\n",
|
||||||
|
" <td>7</td>\n",
|
||||||
|
" <td>North_Carolina</td>\n",
|
||||||
|
" <td>$496 billion</td>\n",
|
||||||
|
" <td>Old North State; Tar Heel State</td>\n",
|
||||||
|
" </tr>\n",
|
||||||
|
" <tr>\n",
|
||||||
|
" <td>8</td>\n",
|
||||||
|
" <td>Pennsylvania</td>\n",
|
||||||
|
" <td>$803 billion</td>\n",
|
||||||
|
" <td>Keystone State; Quaker State</td>\n",
|
||||||
|
" </tr>\n",
|
||||||
|
" <tr>\n",
|
||||||
|
" <td>9</td>\n",
|
||||||
|
" <td>Vermont</td>\n",
|
||||||
|
" <td>None</td>\n",
|
||||||
|
" <td>The Green Mountain State</td>\n",
|
||||||
|
" </tr>\n",
|
||||||
|
" </tbody>\n",
|
||||||
|
"</table>\n",
|
||||||
|
"</div>"
|
||||||
|
],
|
||||||
|
"text/plain": [
|
||||||
|
" stat gsp \\\n",
|
||||||
|
"0 Arizona $259 billion \n",
|
||||||
|
"1 California $3.0 trillion \n",
|
||||||
|
"2 Florida $1.0 trillion \n",
|
||||||
|
"3 Idaho None \n",
|
||||||
|
"4 Maryland $382.4 billion \n",
|
||||||
|
"5 Nevada None \n",
|
||||||
|
"6 New_Jersey None \n",
|
||||||
|
"7 North_Carolina $496 billion \n",
|
||||||
|
"8 Pennsylvania $803 billion \n",
|
||||||
|
"9 Vermont None \n",
|
||||||
|
"\n",
|
||||||
|
" nickname \n",
|
||||||
|
"0 The Grand Canyon State; The Copper State;The V... \n",
|
||||||
|
"1 The Golden State \n",
|
||||||
|
"2 The Sunshine State \n",
|
||||||
|
"3 Gem State \n",
|
||||||
|
"4 Old Line State; Free State; Little America; Am... \n",
|
||||||
|
"5 None \n",
|
||||||
|
"6 The Garden State \n",
|
||||||
|
"7 Old North State; Tar Heel State \n",
|
||||||
|
"8 Keystone State; Quaker State \n",
|
||||||
|
"9 The Green Mountain State "
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"execution_count": 5,
|
||||||
|
"metadata": {},
|
||||||
|
"output_type": "execute_result"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"source": [
|
||||||
|
"res"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"metadata": {
|
||||||
|
"kernelspec": {
|
||||||
|
"display_name": "Python 3",
|
||||||
|
"language": "python",
|
||||||
|
"name": "python3"
|
||||||
|
},
|
||||||
|
"language_info": {
|
||||||
|
"codemirror_mode": {
|
||||||
|
"name": "ipython",
|
||||||
|
"version": 3
|
||||||
|
},
|
||||||
|
"file_extension": ".py",
|
||||||
|
"mimetype": "text/x-python",
|
||||||
|
"name": "python",
|
||||||
|
"nbconvert_exporter": "python",
|
||||||
|
"pygments_lexer": "ipython3",
|
||||||
|
"version": "3.7.1"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"nbformat": 4,
|
||||||
|
"nbformat_minor": 2
|
||||||
|
}
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
stat gsp nickname
|
||||||
|
Arizona $259 billion The Grand Canyon State; The Copper State;The Valentine State
|
||||||
|
California $3.0 trillion The Golden State
|
||||||
|
Florida $1.0 trillion The Sunshine State
|
||||||
|
Idaho None Gem State
|
||||||
|
Maryland $382.4 billion Old Line State; Free State; Little America; America in Miniature
|
||||||
|
Nevada None None
|
||||||
|
New_Jersey None The Garden State
|
||||||
|
North_Carolina $496 billion Old North State; Tar Heel State
|
||||||
|
Pennsylvania $803 billion Keystone State; Quaker State
|
||||||
|
Vermont None The Green Mountain State
|
||||||
|
@@ -0,0 +1,13 @@
|
|||||||
|
Prior to joining the Navy, Hopper earned a Ph.D. in mathematics from Yale University and was a professor of mathematics at Vassar College. Hopper attempted to enlist in the Navy during World War II but was rejected because she was 34 years old. She instead joined the Navy Reserves. Hopper began her computing career in 1944 when she worked on the Harvard Mark I team led by Howard H. Aiken. In 1949, she joined the Eckert–Mauchly Computer Corporation and was part of the team that developed the UNIVAC I computer. At Eckert–Mauchly she began developing the compiler. She believed that a programming language based on English was possible. Her compiler converted English terms into machine code understood by computers.By 1952, Hopper had finished her program linker (originally called a compiler), which was written for the A-0 System. During her wartime service, she co-authored three papers based on her work on the Harvard Mark 1. In 1954, Eckert–Mauchly chose Hopper to lead their department for automatic programming, and she led the release of some of the first compiled languages like FLOW-MATIC. In 1959, she participated in the CODASYL consortium, which consulted Hopper to guide them in creating a machine-independent programming language. This led to the COBOL language, which was inspired by her idea of a language being based on English words.In 1966, she retired from the Naval Reserve, but in 1967 the Navy recalled her to active duty.
|
||||||
|
|
||||||
|
A carbohydrate structural variant of MM glycoprotein (glycophorin A). A variant of the MM glycoprotein (glycophorin A) was isolated from erythrocyte membranes of two individual donors, a mother (L.G.) and daughter (V.W.). This glycoprotein was found to be a carbohydrate variant in which, for both donors, certain O-glycosidically linked saccharides retained the core structure consisting of NeuAc(alpha 2,3)Gal(beta 1,3)GalNAc that is common to all O-linked saccharides of the MN glycoproteins, and, in addition, contained substituents, of varying chain lengths, on the primary carbinol of GalNAc. These saccharides were released from the polypeptide by beta-elimination in the presence of sodium borohydride, and aspects of their structure were investigated by glycosidase digestion and periodate oxidation. Thus, the smallest variant structure was deduced to be NeuAc(alpha 2,3)Gal(beta 1,3)[GlcNAc(beta 1,6)]H2GalNAc. The 6-O-linked GlcNAc appears to serve as the focus of further chain elongation reactions, involving alternate additions of Gal and GlcNAc residues and leading to the formation of several homologous structures. Two such structures, NeuAc(alpha 2,3)Gal(beta 1,3)[GlcNAc(beta 1,?) Gal(beta 1,3/4)GlcNAc(beta 1,6)]H2GalNAc and NeuAc(alpha 2,3) Gal(beta 1,3)[Gal(beta 1,3/4)GlcNAc(beta 1,6)]H2GalNAc were the predominant species present. A larger saccharide was also isolated and its partial sequence was determined to be Gal(beta 1,3/4)GlcNAc(beta 1,?)[Gal(beta 1,3/4)Glc-NAc(beta 1,?)] Gal(beta 1,3/4)GlcNAc(beta 1,6)[NeuAc(alpha 2,3)Gal-(beta 1,3)]H2GalNAc. Because the peptide portion of these glycoproteins contains two methionine residues, it was possible to isolate two CNBr glycopeptides from separate regions of the molecule, and to assess the distribution of these variant structures in the polypeptide.
|
||||||
|
|
||||||
|
Characterization and Antioxidant Activity Determination of Neutral and Acidic Polysaccharides from Panax Ginseng C. A. Meyer. Panax ginseng (P. ginseng) is the most widely consumed herbal plant in Asia and is well-known for its various pharmacological properties. Many studies have been devoted to this natural product. However, polysaccharide's components of ginseng and their biological effects have not been widely studied. In this study, white ginseng neutral polysaccharide (WGNP) and white ginseng acidic polysaccharide (WGAP) fractions were purified from P. ginseng roots. The chemical properties of WGNP and WGAP were investigated using various chromatography and spectroscopy techniques, including high-performance gel permeation chromatography, Fourier-transform infrared spectroscopy, and high-performance liquid chromatography with an ultra-violet detector. The antioxidant, anti-radical, and hydrogen peroxide scavenging activities were evaluated in vitro and in vivo using Caenorhabditis elegans as the model organism. Our in vitro data by ABTS (2,2'-azino-bis-(3-ethylbenzothiazoline-6-sulfonic acid), reducing power, ferrous ion chelating, and hydroxyl radical scavenging activity suggested that the WGAP with significantly higher uronic acid content and higher molecular weight exhibits a much stronger antioxidant effect as compared to that of WGNP. Similar antioxidant activity of WGAP was also confirmed in vivo by evaluating internal reactive oxygen species (ROS) concentration and lipid peroxidation. In conclusion, WGAP may be used as a natural antioxidant with potent scavenging and metal chelation properties.
|
||||||
|
|
||||||
|
Protein Thermodynamic Destabilization in the Assessment of Pathogenicity of a Variant of Uncertain Significance in Cardiac Myosin Binding Protein C. In the era of next generation sequencing (NGS), genetic testing for inherited disorders identifies an ever-increasing number of variants whose pathogenicity remains unclear. These variants of uncertain significance (VUS) limit the reach of genetic testing in clinical practice. The VUS for hypertrophic cardiomyopathy (HCM), the most common familial heart disease, constitute over 60% of entries for missense variants shown in ClinVar database. We have studied a novel VUS (c.1809T>G-p.I603M) in the most frequently mutated gene in HCM, MYBPC3, which codes for cardiac myosin-binding protein C (cMyBPC). Our determinations of pathogenicity integrate bioinformatics evaluation and functional studies of RNA splicing and protein thermodynamic stability. In silico prediction and mRNA analysis indicated no alteration of RNA splicing induced by the variant. At the protein level, the p.I603M mutation maps to the C4 domain of cMyBPC. Although the mutation does not perturb much the overall structure of the C4 domain, the stability of C4 I603M is severely compromised as detected by circular dichroism and differential scanning calorimetry experiments. Taking into account the highly destabilizing effect of the mutation in the structure of C4, we propose reclassification of variant p.I603M as likely pathogenic. Looking into the future, the workflow described here can be used to refine the assignment of pathogenicity of variants of uncertain significance in MYBPC3.
|
||||||
|
|
||||||
|
The idea of neural language models as introduced by Bengio et al. [5] is to jointly learn an em- bedding of words into an n-dimensional vector space and to use these vectors to predict how likely a word is given its context. Collobert and Weston [6] introduced a new neural network model to compute such an embedding. When these networks are optimized via gradient ascent the derivatives modify the word embedding matrix L ∈ Rn×|V |, where |V | is the size of the vocabulary. The word vectors inside the embedding matrix capture distributional syntactic and semantic information via the word’s co-occurrence statistics. For further details and evaluations of these embeddings, see [5, 6, 7, 8]. Once this matrix is learned on an unlabeled corpus, we can use it for subsequent tasks by using each word’s vector (a column in L) to represent that word. In the remainder of this paper, we represent a sentence (or any n-gram) as an ordered list of these vectors (x1 , . . . , xm ). This word representation is better suited for autoencoders than the binary number representations used in previous related autoencoder models such as the recursive autoassociative memory (RAAM) model of Pollack [9, 10] or recurrent neural networks [11] since the activations are inherently continuous.
|
||||||
|
|
||||||
|
The sequences mediating receptor mRNA down-regulation are represented within the AR cDNA and not within the CMV promoter. Androgenic down-regulation of AR cDNA expression was time- and dose-dependent, resembling native AR mRNA down-regulation. In addition, androgenic regulation of the receptor cDNA was not dependent on protein synthesis suggesting that AR and/or another pre-existing protein(s) is involved in this process. In COS 1 cells co-transfected with androgen and glucocorticoid receptor cDNAs, dexamethasone mimicked the action of androgen in down-regulating AR mRNA. This response depended on glucocorticoid receptors. Androgen had little effect on steady-state levels of AR protein consistent with reports that androgen down-regulates AR mRNA but increases AR protein half-life (Kemppainen et al. (1992) J. Biol. Chem. 267, 968-974; Zhou et al. (1995) Mol. Endocrinol. 9, 208-218). However, glucocorticoids decreased AR protein levels in cells that co-expressed androgen and glucocorticoid receptors. These results indicate that sequences represented in the AR cDNA mediate AR mRNA down-regulation by both androgens and glucocorticoids. Inhibition of AR mRNA and protein by glucocorticoids suggests that these steroids may modulate androgen action in tissues, such as mammary gland and prostate, which express both androgen and glucocorticoid receptors.
|
||||||
|
|
||||||
|
Immunohistochemical application of antibodies against heparan sulfate proteoglycan core protein and heparitinase-digested heparan sulfate stubs showed the presence of heparan sulfate proteoglycan in all basement membranes of the rat kidney. However, a monoclonal antibody (JM-403) against native heparan sulfate (van den Born, J., van den Heuvel, L. P. W. J., Bakker, M. A. H., Veerkamp, J. H., Assmann, K. J. M., and Berden, J. H. M. (1992) Kidney Int. 41, 115-123) largely failed to stain tubular basement membranes, suggesting the presence of heparan sulfate chains lacking the specific JM-403 epitope. Heparan sulfate preparations from various sources differed markedly with regard to JM-403 binding, as demonstrated by liquid phase inhibition in enzyme-linked immunosorbent assay, the interaction decreasing with increasing sulfate contents of the polysaccharide. Mapping of the JM-403 epitope indicated that it was dominated by one or more N-unsubstituted glucosamine unit(s), since treatments that destroyed or altered the structure of such units in heparan sulfate preparations (cleavage at N-unsubstituted glucosamine units with HNO2 at pH 3.9 and N-acetylation with acetic anhydride, respectively), abolished antibody binding.
|
||||||
|
After Width: | Height: | Size: 179 KiB |
@@ -0,0 +1,108 @@
|
|||||||
|
review_date,star_rating,count
|
||||||
|
2015-01-02,2,1
|
||||||
|
2015-01-02,3,2
|
||||||
|
2015-01-02,4,1
|
||||||
|
2015-01-02,5,6
|
||||||
|
2015-01-03,1,4
|
||||||
|
2015-01-03,2,1
|
||||||
|
2015-01-03,3,3
|
||||||
|
2015-01-03,4,3
|
||||||
|
2015-01-03,5,9
|
||||||
|
2015-01-04,1,1
|
||||||
|
2015-01-04,2,2
|
||||||
|
2015-01-04,3,1
|
||||||
|
2015-01-04,4,3
|
||||||
|
2015-01-04,5,8
|
||||||
|
2015-01-05,1,4
|
||||||
|
2015-01-05,3,2
|
||||||
|
2015-01-05,4,6
|
||||||
|
2015-01-05,5,14
|
||||||
|
2015-01-06,2,1
|
||||||
|
2015-01-06,3,3
|
||||||
|
2015-01-06,4,6
|
||||||
|
2015-01-06,5,8
|
||||||
|
2015-01-07,1,1
|
||||||
|
2015-01-07,2,4
|
||||||
|
2015-01-07,3,1
|
||||||
|
2015-01-07,4,5
|
||||||
|
2015-01-07,5,8
|
||||||
|
2015-01-08,1,1
|
||||||
|
2015-01-08,3,2
|
||||||
|
2015-01-08,4,2
|
||||||
|
2015-01-08,5,10
|
||||||
|
2015-01-09,1,2
|
||||||
|
2015-01-09,2,2
|
||||||
|
2015-01-09,3,2
|
||||||
|
2015-01-09,4,3
|
||||||
|
2015-01-09,5,20
|
||||||
|
2015-01-10,1,1
|
||||||
|
2015-01-10,3,2
|
||||||
|
2015-01-10,4,1
|
||||||
|
2015-01-10,5,4
|
||||||
|
2015-01-11,1,1
|
||||||
|
2015-01-11,3,1
|
||||||
|
2015-01-11,4,3
|
||||||
|
2015-01-11,5,10
|
||||||
|
2015-01-12,1,3
|
||||||
|
2015-01-12,2,1
|
||||||
|
2015-01-12,3,1
|
||||||
|
2015-01-12,4,4
|
||||||
|
2015-01-12,5,9
|
||||||
|
2015-01-13,3,1
|
||||||
|
2015-01-13,4,2
|
||||||
|
2015-01-13,5,15
|
||||||
|
2015-01-14,1,1
|
||||||
|
2015-01-14,3,3
|
||||||
|
2015-01-14,5,9
|
||||||
|
2015-01-15,1,1
|
||||||
|
2015-01-15,3,1
|
||||||
|
2015-01-15,4,3
|
||||||
|
2015-01-15,5,8
|
||||||
|
2015-01-16,3,3
|
||||||
|
2015-01-16,4,8
|
||||||
|
2015-01-16,5,5
|
||||||
|
2015-01-17,4,3
|
||||||
|
2015-01-17,5,6
|
||||||
|
2015-01-18,1,3
|
||||||
|
2015-01-18,2,2
|
||||||
|
2015-01-18,3,1
|
||||||
|
2015-01-18,4,1
|
||||||
|
2015-01-18,5,14
|
||||||
|
2015-01-19,1,1
|
||||||
|
2015-01-19,4,1
|
||||||
|
2015-01-19,5,4
|
||||||
|
2015-01-20,1,1
|
||||||
|
2015-01-20,3,3
|
||||||
|
2015-01-20,4,5
|
||||||
|
2015-01-20,5,14
|
||||||
|
2015-01-21,1,2
|
||||||
|
2015-01-21,2,1
|
||||||
|
2015-01-21,3,2
|
||||||
|
2015-01-21,4,1
|
||||||
|
2015-01-21,5,14
|
||||||
|
2015-01-22,1,1
|
||||||
|
2015-01-22,5,10
|
||||||
|
2015-01-23,2,1
|
||||||
|
2015-01-23,3,1
|
||||||
|
2015-01-23,5,3
|
||||||
|
2015-01-24,3,2
|
||||||
|
2015-01-24,4,1
|
||||||
|
2015-01-24,5,5
|
||||||
|
2015-01-25,1,1
|
||||||
|
2015-01-25,2,1
|
||||||
|
2015-01-25,4,3
|
||||||
|
2015-01-26,2,2
|
||||||
|
2015-01-26,5,8
|
||||||
|
2015-01-27,5,9
|
||||||
|
2015-01-28,3,2
|
||||||
|
2015-01-28,4,1
|
||||||
|
2015-01-28,5,6
|
||||||
|
2015-01-29,4,3
|
||||||
|
2015-01-29,5,16
|
||||||
|
2015-01-30,1,1
|
||||||
|
2015-01-30,2,1
|
||||||
|
2015-01-30,5,5
|
||||||
|
2015-01-31,2,2
|
||||||
|
2015-01-31,3,2
|
||||||
|
2015-01-31,4,5
|
||||||
|
2015-01-31,5,8
|
||||||
|
|
After Width: | Height: | Size: 35 KiB |
|
After Width: | Height: | Size: 32 KiB |