ページ

ラベル 速度 の投稿を表示しています。 すべての投稿を表示
ラベル 速度 の投稿を表示しています。 すべての投稿を表示

2016年5月7日土曜日

Mathematica Benchimark

Benchmarkingパッケージを使ったMathematicaのベンチマーク

私物laptop
Intel(R) Core(TM) i7-5500U CPU @ 2.40GHz
# Core: 2
# Threading: 4
{"MachineName" -> "--", "System" -> "Linux x86 (64-bit)",
"BenchmarkName" -> "WolframMark", "FullVersionNumber" -> "10.3.1",
"Date" -> "January 31, 2016", "BenchmarkResult" -> 1.392,
"TotalTime" -> 9.945, "Results" -> {{"Data Fitting", 0.537},
{"Digits of Pi", 0.341}, {"Discrete Fourier Transform", 0.68},
{"Eigenvalues of a Matrix", 0.608}, {"Elementary Functions", 0.567},
{"Gamma Function", 0.467}, {"Large Integer Multiplication", 0.467},
{"Matrix Arithmetic", 0.334}, {"Matrix Multiplication", 0.754},
{"Matrix Transpose", 1.61}, {"Numerical Integration", 0.676},
{"Polynomial Expansion", 0.092}, {"Random Number Sort", 1.16},
{"Singular Value Decomposition", 0.765}, {"Solving a Linear System", 0.887}}}

------------------------------------------------------------------------------------------------
とあるサーバー1:
Intel(R) Xeon(R) CPU X5550 @ 2.67GHz
# Core: 4
# Threading: 8
{"MachineName" -> "--", "System" -> "Linux x86 (64-bit)",
"BenchmarkName" -> "WolframMark", "FullVersionNumber" -> "10.0.2",
"Date" -> "January 31, 2016", "BenchmarkResult" -> 1.332,
"TotalTime" -> 10.388, "Results" -> {{"Data Fitting", 1.059},
{"Digits of Pi", 0.507}, {"Discrete Fourier Transform", 0.545},
{"Eigenvalues of a Matrix", 1.001}, {"Elementary Functions", 0.389},
{"Gamma Function", 0.66}, {"Large Integer Multiplication", 0.668},
{"Matrix Arithmetic", 0.536}, {"Matrix Multiplication", 0.408},
{"Matrix Transpose", 1.022}, {"Numerical Integration", 1.002},
{"Polynomial Expansion", 0.136}, {"Random Number Sort", 1.159},
{"Singular Value Decomposition", 0.651}, {"Solving a Linear System", 0.645}}}

------------------------------------------------------------------------------------------------
とあるサーバー2:
Intel(R) Xeon(R) CPU E5607 @ 2.27GHz
# Core: 4?8?
# Threading: 8
{"MachineName" -> "--", "System" -> "Linux x86 (64-bit)",
"BenchmarkName" -> "WolframMark", "FullVersionNumber" -> "10.0.2",
"Date" -> "January 31, 2016", "BenchmarkResult" -> 1.056,
"TotalTime" -> 13.103, "Results" -> {{"Data Fitting", 0.962},
{"Digits of Pi", 0.648}, {"Discrete Fourier Transform", 0.6},
{"Eigenvalues of a Matrix", 1.076}, {"Elementary Functions", 0.481},
{"Gamma Function", 0.874}, {"Large Integer Multiplication", 0.916},
{"Matrix Arithmetic", 0.706}, {"Matrix Multiplication", 0.541},
{"Matrix Transpose", 1.485}, {"Numerical Integration", 1.342},
{"Polynomial Expansion", 0.14}, {"Random Number Sort", 1.542},
{"Singular Value Decomposition", 0.878}, {"Solving a Linear System", 0.912}}}

------------------------------------------------------------------------------------------------
Model Name: iMac
Processor Name: Intel Core i7
Processor Speed: 3.1 GHz
# Core: 4
# Threading: 8

{"MachineName" -> "--", "System" -> "Mac OS X x86 (64-bit)",
"BenchmarkName" -> "WolframMark", "FullVersionNumber" -> "10.3.1",
"Date" -> "January 31, 2016", "BenchmarkResult" -> 1.985,
"TotalTime" -> 6.975, "Results" -> {{"Data Fitting", 0.689},
{"Digits of Pi", 0.295}, {"Discrete Fourier Transform", 0.369},
{"Eigenvalues of a Matrix", 0.474}, {"Elementary Functions", 0.506},
{"Gamma Function", 0.353}, {"Large Integer Multiplication", 0.33},
{"Matrix Arithmetic", 0.632}, {"Matrix Multiplication", 0.259},
{"Matrix Transpose", 0.726}, {"Numerical Integration", 0.62},
{"Polynomial Expansion", 0.086}, {"Random Number Sort", 0.819},
{"Singular Value Decomposition", 0.435}, {"Solving a Linear System", 0.382}}}

------------------------------------------------------------------------------------------------
Model Name: iMac
Processor Name: Intel Core 2 Duo
Processor Speed: 3.06 GHz
# Core: 2
# Threading: 2(?)

{"MachineName" -> "--", "System" -> "Mac OS X x86 (64-bit)",
"BenchmarkName" -> "WolframMark", "FullVersionNumber" -> "10.0.2",
"Date" -> "January 31, 2016", "BenchmarkResult" -> 0.613,
"TotalTime" -> 22.586, "Results" -> {{"Data Fitting", 1.713},
{"Digits of Pi", 0.67}, {"Discrete Fourier Transform", 1.548},
{"Eigenvalues of a Matrix", 1.356}, {"Elementary Functions", 2.113},
{"Gamma Function", 0.768}, {"Large Integer Multiplication", 0.769},
{"Matrix Arithmetic", 2.632}, {"Matrix Multiplication", 1.646},
{"Matrix Transpose", 1.804}, {"Numerical Integration", 1.949},
{"Polynomial Expansion", 0.267}, {"Random Number Sort", 1.669},
{"Singular Value Decomposition", 1.992}, {"Solving a Linear System", 1.69}}}

-----------------------------------------------------------------------------------------------
raspberry pi 2
Coretex-A7
# Core: 4
# threading: 4(?)
{"MachineName" -> "--", "System" -> "Linux ARM (32-bit)",
"BenchmarkName" -> "WolframMark", "FullVersionNumber" -> "10.0.2",
"Date" -> "January 24, 2016", "BenchmarkResult" -> 0.029,
"TotalTime" -> 475.891,
"Results" -> {{"Data Fitting", 10.904}, {"Digits of Pi", 4.805}, {
"Discrete Fourier Transform", 30.977}, {
"Eigenvalues of a Matrix", 35.914}, {
"Elementary Functions", 25.479}, {"Gamma Function", 6.437}, {
"Large Integer Multiplication", 6.735}, {
"Matrix Arithmetic", 6.07}, {"Matrix Multiplication", 151.389}, {
"Matrix Transpose", 7.202}, {"Numerical Integration", 11.684}, {
"Polynomial Expansion", 1.086}, {"Random Number Sort", 8.376}, {
"Singular Value Decomposition", 75.031}, {
"Solving a Linear System", 93.802}}}

----------------------------------------------------------------------------------------------------
とあるサーバー

Intel(R) Xeon(R) CPU E5-1680 v3 @ 3.20GHz

{"MachineName" -> "--", "System" -> "Linux x86 (64-bit)",
 "BenchmarkName" -> "WolframMark", "FullVersionNumber" -> "10.4.1",
 "Date" -> "June 1, 2016", "BenchmarkResult" -> 2.06, "TotalTime" -> 6.721,
 "Results" -> {{"Data Fitting", 0.568}, {"Digits of Pi", 0.293},
   {"Discrete Fourier Transform", 0.274}, {"Eigenvalues of a Matrix",
    0.78}, {"Elementary Functions", 0.209}, {"Gamma Function", 0.395},
   {"Large Integer Multiplication", 0.389}, {"Matrix Arithmetic", 0.214},
   {"Matrix Multiplication", 0.282}, {"Matrix Transpose", 0.611},
   {"Numerical Integration", 0.705}, {"Polynomial Expansion", 0.098},
   {"Random Number Sort", 0.984}, {"Singular Value Decomposition", 0.589},
   {"Solving a Linear System", 0.33}}}

2016年5月6日金曜日

fortran 4倍精度で,計算速度はどの程度遅くなるか?

大学の環境で利用可能なコンパイラー
gfortran, ifort (intel), xl fortran (IBM), PGI fortran (NVIDIA)で
倍精度から4倍精度にした際にどの程度計算速度が遅くなるか比較する.
なお pgi fortranでは4倍精度のプログラムがコンパイルできなかったので,
実際には残りの3つの比較.

計算内容は割り算と掛け算をひたすら繰り返しただけ.

最適化(オプション)なしの比較.
----------------------- 結果1 (intel cpu)--------------------------------------

gfortran quadruple/double ratio
real: 9.6 倍遅延
complex : 19倍遅延

ifort quadruple/double ratio
real: 16倍 遅延
complex: 28倍遅延

efficiency (gfortran calc. time/ ifort calc. time)
gfortran -> ifort
real: 5.988sec/4.963sec -> 1.2倍 高速化
complex 22sec/16sec  ->1.3倍 高速化
(ifortの方がちょっとだけ早い)

--------------------------結果2 (ibm ワークステーション)------------------------

xlf90 quadruple/double ratio
real: 2.27倍遅延
complex: 5.83倍遅延

efficency  (gfortran -> xlf90)
gfortran 4.7.3で4倍精度コンパイルできなかったので比較できず.
(4.6から対応だと思っていたが・・・)

pgf90 4倍精度コンパイルできず...

-----------------------------------------------------------------------------
上記比較はオプションなしの場合.

ifortは倍精度は2-3倍オプションなしで早くなるが,
4倍精度では思ったより実力を発揮しなかった.
恐らく4倍精度はソフトウェア上での実装なので,最適化があまり意味がないのかと思う.
ただし,ifortの場合do文を自動並列化ができるので強制的に並列化してやれば,最大でcpu(thread)数倍くらい早くならないこともない.


ibmのコンパイラーはibmのマシーンに特化してる気もする.
他にもNAGや富士通(販売終了),absoftがあるらしいが,使ったこと無い.

価格2016年5月時点
IBM: $3380 
NAG: (63,000円) single user license 
Intel: 10-30万くらい (Parallel Studio XE Composer Edition)
PGI:5-10万くらい
absoft:6万-12万 
備考:intel compiler の非商用版(linuxのみ)は無償だったが,有償版だけになった模様.
学生に限りは無償使用のライセンスが在る模様.

2013年11月21日木曜日

Maximaによる任意精度計算

Mathematicaはいつ使えなくなるか(ライセンス的な関係で)分らないので,Maximaを任意精度数値計算に使えないか試してみる.


精度(precision)のコントロールは

fpprec:50

演算による多倍長浮動小数点の有理数への変換を制御するため

bftorat: true

にする必要がある.デフォルトではfalse.
falseの場合多倍長浮動小数点は小さい有理数で近似される.

また,出力する桁のコントロールは

fpprintprec:50;

で可能.
maxima 5.30.0 だとload("fft")が出来なかったので,古いバージョンを使う.
参考:http://maxima.sourceforge.jp/maxima_5.html
$ maxima
Maxima 5.21post http://maxima.sourceforge.net
using Lisp SBCL 1.0.39
Distributed under the GNU Public License. See the file COPYING.
Dedicated to the memory of William Schelter.
The function bug_report() provides bug reporting information.
(%i1) fpprec;
(%o1) 16
(%i2) fpprec:50;
(%o2) 50
(%i5) bftorat:true;
(%o5) true
(%i6) x:bfloat([0,1/3,0,0]);
(%o6) [0.0b0,3.3333333333333333333333333333333333333333333333333b−1,0.0b0,0.0b0]
(%i7) load("fft");
(%o7) "/Applications/Maxima.app/share/maxima/5.21post/share/numeric/fft.lisp"
(%i10) res:inverse_fft(fft(x));
(%o10) [0.0,.3333333333333333,0.0,0.0]
(%i11) bfloatp(res[1]);
(%o11) false

FFTは任意精度計算やってくれないっぽい.一方で行列mat=((i,pi),(pi,i))の対角化は任意精度でやってくれるっぽい.上のmatの固有値は(±pi, i)である.

(%i12) mat:bfloat(matrix([%i,%pi],[%pi,%i]));
(%o12) matrix([%i,3.1415926535897932384626433832795028841971693993751b0],[3.1415926535897932384626433832795028841971693993751b0,%i])
(%i13) load("eigen");
(%o13) "/Applications/Maxima.app/share/maxima/5.21post/share/matrix/eigen.mac"
(%i14) res:eigenvalues(mat);
`rat' replaced -9.8696044010893586188344909998761511353136994072408B0 by -41494057121218467653886768/4204226981644629723547549 = -9.8696044010893586188344909998761511353136994072408B0
(%o14) [[(4204226981644629723547549*%i−4*sqrt(10903152157933135742303986703335184117163956870727))/4204226981644629723547549,(4204226981644629723547549*%i+4*sqrt(10903152157933135742303986703335184117163956870727))/4204226981644629723547549],[1,1]]
(%i24) bfloat(realpart(res[1][1])); imagpart(res[1][1]);
(%o24) −3.1415926535897932384626433832795028841971693993751b0
(%o25) 1
(%i22) -bfloat(%pi);
(%o22) −3.1415926535897932384626433832795028841971693993751b0
(%i26) bfloat(realpart(res[1][2])); imagpart(res[1][2]);
(%o26) +3.1415926535897932384626433832795028841971693993751b0
(%o27) 1
---------- Mathmatica res ----------
In[2]:= N[Pi, 50]
Out[2]= 3.1415926535897932384626433832795028841971693993751

maxima対角化は時間が掛かりすぎるから現実的ではない・・・か
(%i1) f[i,j]:=random(1.0);
(%o1) f[i,j]:=random(1.0)
(%i6) showtime:true;
Evaluation took 0.0000 seconds (0.0000 elapsed) using 0 bytes.
(%o6) true
(%i12) mat:genmatrix(f,4,4);
Evaluation took 0.0000 seconds (0.0000 elapsed) using 0 bytes.
(%o12) matrix([.9138095996128959,.1951846177977887,.4823905248516196,.3554400571592862],[.4155627227214829,.6125738892538246,.8418126553359213,.5055894446438118],[0.0363933158999219,.6252975183418072,.9937314578694818,.04158924615444981],[.5931521815169765,.6546577661145074,.4707106740336644,.05192355366451751])
(%i13) eigenvalues(mat);
Evaluation took 9.2310 seconds (14.0640 elapsed) using 2764.567 MB.

2013年10月27日日曜日

mpmath 速度比べ

import mpmath
import numpy as np
import time

sample=500
iter=500

x = np.array([1.0/3.0]*sample)
init=x[0]
start = time.time()
for i in range(iter):
    if i<iter/2:
        x = x + 1.0/3.0
    else:
        x = x - 1.0/3.0
goal=time.time()
print("---- numpy array -----")
print("%.20f" % init)
print("%.20f" % x[0])
print("time:", goal-start)

mpmath.mp.dps = 18
x = np.array([mpmath.mpf("1.0")/mpmath.mpf("3.0")]*sample)
init=x[0]
start = time.time()
for i in range(iter):
    if i<iter/2:
        x = x + mpmath.mpf("1.0")/mpmath.mpf("3.0")
    else:
        x = x - mpmath.mpf("1.0")/mpmath.mpf("3.0")
goal=time.time()
print("---- mpmath with numpy array (dps=%d) -----" % mpmath.mp.dps)
print("%.20s" % init)
print("%.20s" % x[0])
print("time:", goal-start)

mpmath.mp.dps = 100
x = np.array([mpmath.mpf("1.0")/mpmath.mpf("3.0")]*sample)
init=x[0]
start = time.time()
for i in range(iter):
    if i<iter/2:
        x = x + mpmath.mpf("1.0")/mpmath.mpf("3.0")
    else:
        x = x - mpmath.mpf("1.0")/mpmath.mpf("3.0")
goal=time.time()
print("---- mpmath with numpy array (dps=%d) -----" % mpmath.mp.dps)
print("%.20s" % init)
print("%.20s" % x[0])
print("time:", goal-start)

mpmath.mp.dps = 18
x = mpmath.matrix([mpmath.mpf("1.0")/mpmath.mpf("3.0")]*sample)
init=x[0]
start = time.time()
for i in range(iter):
    if i<iter/2:
        x = x + mpmath.mpf("1.0")/mpmath.mpf("3.0")
    else:
        x = x - mpmath.mpf("1.0")/mpmath.mpf("3.0")
goal=time.time()
print("---- mpmath matrix (dps=%d)-----" % mpmath.mp.dps)
print("%.20s" % init)
print("%.20s" % x[0])
print("time:", goal-start)

mpmath.mp.dps = 100
x = mpmath.matrix([mpmath.mpf("1.0")/mpmath.mpf("3.0")]*sample)
init=x[0]
start = time.time()
for i in range(iter):
    if i<iter/2:
        x = x + mpmath.mpf("1.0")/mpmath.mpf("3.0")
    else:
        x = x - mpmath.mpf("1.0")/mpmath.mpf("3.0")
goal=time.time()
print("---- mpmath matrix (dps=%d)-----" % mpmath.mp.dps)
print("%.20s" % init)
print("%.20s" % x[0])
print("time:", goal-start)


x = [mpmath.mpf("1.0")/mpmath.mpf("3.0")]*sample
init=x[0]
start = time.time()
for i in range(iter):
    if i<iter/2:
        x = [xx + mpmath.mpf("1.0")/mpmath.mpf("3.0") for xx in x]
    else:
        x = [xx - mpmath.mpf("1.0")/mpmath.mpf("3.0") for xx in x]
goal=time.time()
print("---- list -----")
print("%.20f" % init)
print("%.20f" % x[0])
print("time:", goal-start)




実行結果
$ python speed_test_array.py 
---- numpy array -----
0.33333333333333331483
0.33333333333332632042
('time:', 0.0022330284118652344)
---- mpmath with numpy array (dps=18) -----
0.333333333333333333
0.333333333333333326
('time:', 1.7795300483703613)
---- mpmath with numpy array (dps=100) -----
0.333333333333333333
0.333333333333333333
('time:', 2.18923282623291)
---- mpmath matrix (dps=18)-----
0.333333333333333333
0.333333333333333326
('time:', 5.815176010131836)
---- mpmath matrix (dps=100)-----
0.333333333333333333
0.333333333333333333
('time:', 6.364114046096802)
---- list -----
0.33333333333333331483
0.33333333333333331483
('time:', 12.696530103683472)